feat(milestone): merge phase/01 mastery-core → milestone/v0.3-mastery-scoring
Phase 1 complete. Mastery scoring + competency rubrics + VC issuer shipped. 9 slices, 5 waves, 238 tests passing, 13/13 REQ-IDs covered. 4/4 grill MUST conditions satisfied. VERIFY: APPROVE_WITH_NOTES. ---ci--- project: praxis phase: 1 milestone: v0.3 status: complete requirements: covered: [REQ-MAST-01, REQ-MAST-02, REQ-MAST-03, REQ-SCEN-02, REQ-SCEN-03, REQ-SCEN-04, REQ-PATH-02, REQ-NFR-MAST-01, REQ-NFR-MAST-02, REQ-NFR-VC-01, REQ-NFR-VC-02, REQ-NFR-IRT-01] partial: [] ---/ci---
This commit is contained in:
@@ -0,0 +1,316 @@
|
||||
#!/usr/bin/env python3
|
||||
"""SLICE-08 TASK-08-01 — End-to-end P1 mastery smoke test (not a pytest).
|
||||
|
||||
Simulates 3 sessions across 3 distinct Customer-Service scenarios → runs the
|
||||
mastery flow (with a mocked LLM returning canned verbatim-quote evidence) →
|
||||
verifies:
|
||||
- mastery gate opens after the 3rd passing scenario with path score >= 3.5
|
||||
- theta converges upward (passes against increasing difficulty)
|
||||
- progress advances week-by-week as each week's gate opens
|
||||
- one mastery_gate_event row is recorded per session in SQLite
|
||||
|
||||
Runnable: `python3 scripts/test_mastery_e2e.py`
|
||||
Exit code 0 on PASS, 1 on FAIL. Prints a PASS/FAIL summary.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from db.store import PraxisStore, HARDCODED_LEARNER_ID
|
||||
from server.mastery.irt import IRTEngine, DEFAULT_THETA
|
||||
from server.mastery.rubric_loader import clear_cache, load_rubric
|
||||
from server.paths.engine import PathEngine
|
||||
from server.scenarios.loader import load as load_scenario
|
||||
from server.session_recorder import MasteryFlowDeps, SessionRecorder
|
||||
|
||||
_REPO = Path(__file__).resolve().parent.parent
|
||||
_RUBRICS_DIR = _REPO / "rubrics"
|
||||
_SCENARIOS_DIR = _REPO / "scenarios"
|
||||
_PATHS_DIR = _REPO / "paths"
|
||||
|
||||
_PATH_SLUG = "customer_service"
|
||||
_SCENARIO_IDS = [
|
||||
"cs_refund_ca_v01",
|
||||
"cs_escalation_ca_v02",
|
||||
"cs_policy_exception_ca_v03",
|
||||
]
|
||||
|
||||
|
||||
def _transcript_for(scenario_id: str) -> list[dict[str, str]]:
|
||||
if scenario_id == "cs_refund_ca_v01":
|
||||
learner_a = (
|
||||
"I'm really sorry the bowl arrived cracked — that's genuinely "
|
||||
"frustrating. I can refund the full amount to your original card "
|
||||
"within 3 business days, or send a replacement first class tomorrow. "
|
||||
"Which would you prefer?"
|
||||
)
|
||||
learner_b = (
|
||||
"Of course — I've issued a full refund of $42.99 to your Visa ending "
|
||||
"4421. You'll see it in 2-3 business days. Is there anything else I "
|
||||
"can help with today?"
|
||||
)
|
||||
elif scenario_id == "cs_escalation_ca_v02":
|
||||
learner_a = (
|
||||
"I hear you — two weeks with no straight answers is genuinely "
|
||||
"infuriating, and you're right to push for clarity. I'm not going to "
|
||||
"hide behind policy. Here's what I can do right now: I'll trace the "
|
||||
"shipment, refund the shipping cost today, and give you a firm "
|
||||
"delivery date within 24 hours. Would that work?"
|
||||
)
|
||||
learner_b = (
|
||||
"Thank you for staying with me on this. I've refunded the $9.50 "
|
||||
"shipping charge to your card and flagged the order for immediate "
|
||||
"dispatch. You'll get a tracking number by email within the hour. "
|
||||
"Is there anything else I can do for you?"
|
||||
)
|
||||
else:
|
||||
learner_a = (
|
||||
"You're absolutely right — a defect appearing last week is a "
|
||||
"different situation from a 45-day change-of-mind. The 30-day window "
|
||||
"is a guideline for returns, not a hard wall for defects. I can "
|
||||
"offer a partial credit of 70% toward a replacement, or start a "
|
||||
"manufacturer warranty claim on your behalf. Which would you prefer?"
|
||||
)
|
||||
learner_b = (
|
||||
"I've issued a $30 partial credit to your original payment method "
|
||||
"and started the manufacturer warranty claim — they'll reach out "
|
||||
"within 5 business days. You'll get a confirmation email within the "
|
||||
"hour. Anything else I can help with today?"
|
||||
)
|
||||
return [
|
||||
{"role": "customer", "content": "I'm upset and need this resolved now."},
|
||||
{"role": "learner", "content": learner_a},
|
||||
{"role": "customer", "content": "Okay, go ahead with that."},
|
||||
{"role": "learner", "content": learner_b},
|
||||
]
|
||||
|
||||
|
||||
def _canned_evidence(transcript: list[dict[str, str]]) -> str:
|
||||
t1 = transcript[1]["content"]
|
||||
t2 = transcript[3]["content"]
|
||||
return json.dumps(
|
||||
[
|
||||
{
|
||||
"criterion_id": "empathy",
|
||||
"quote": t1,
|
||||
"signals": [
|
||||
"named_emotion_in_own_words",
|
||||
"acknowledged_specific",
|
||||
"tone_pace_adjusted",
|
||||
"multiple_acknowledgement_instances",
|
||||
],
|
||||
},
|
||||
{
|
||||
"criterion_id": "resolution",
|
||||
"quote": t1,
|
||||
"signals": [
|
||||
"concrete_method",
|
||||
"concrete_amount_or_channel",
|
||||
"concrete_next_step",
|
||||
"decision_tree_of_options",
|
||||
"matched_to_customer_preference",
|
||||
"confirms_acceptance",
|
||||
],
|
||||
},
|
||||
{
|
||||
"criterion_id": "de_escalation",
|
||||
"quote": t1,
|
||||
"signals": [
|
||||
"explicit_acknowledge_reframe_offer",
|
||||
"cycles_acknowledge_reframe",
|
||||
"lowers_intensity_without_conceding_policy",
|
||||
],
|
||||
},
|
||||
{
|
||||
"criterion_id": "professionalism",
|
||||
"quote": t2,
|
||||
"signals": [
|
||||
"plain_language",
|
||||
"in_role_throughout",
|
||||
"no_prohibited_advice",
|
||||
"adapts_register",
|
||||
"concise_for_voice",
|
||||
"manages_silence",
|
||||
],
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class _ScriptedLLM:
|
||||
def __init__(self, raws: list[str]) -> None:
|
||||
self._iter = iter(raws)
|
||||
|
||||
async def chat_full(
|
||||
self,
|
||||
messages: list[dict[str, str]],
|
||||
*,
|
||||
model: str | None = None,
|
||||
no_think: bool = False,
|
||||
) -> tuple[str, dict[str, Any]]:
|
||||
try:
|
||||
raw = next(self._iter)
|
||||
except StopIteration as exc:
|
||||
raise RuntimeError("scripted LLM exhausted") from exc
|
||||
return raw, {"model": model or "test"}
|
||||
|
||||
|
||||
def _deps(llm: Any, scenario_id: str) -> MasteryFlowDeps:
|
||||
clear_cache()
|
||||
return MasteryFlowDeps(
|
||||
llm=llm,
|
||||
irt=IRTEngine(),
|
||||
path_engine=PathEngine(paths_dir=_PATHS_DIR),
|
||||
load_rubric=lambda: load_rubric(_PATH_SLUG, rubrics_dir=_RUBRICS_DIR),
|
||||
load_scenario=lambda: load_scenario(scenario_id, scenarios_dir=_SCENARIOS_DIR),
|
||||
load_path=lambda: PathEngine(paths_dir=_PATHS_DIR).load_path(_PATH_SLUG),
|
||||
)
|
||||
|
||||
|
||||
def _fmt_pass(label: str) -> str:
|
||||
return f" PASS {label}"
|
||||
|
||||
|
||||
def _fmt_fail(label: str, detail: str) -> str:
|
||||
return f" FAIL {label} — {detail}"
|
||||
|
||||
|
||||
async def _run() -> int:
|
||||
failures: list[str] = []
|
||||
print("=" * 70)
|
||||
print("SLICE-08 TASK-08-01 — End-to-end P1 mastery smoke test")
|
||||
print("=" * 70)
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="praxis_e2e_") as tmp:
|
||||
db_path = Path(tmp) / "e2e.db"
|
||||
store = PraxisStore(db_path)
|
||||
await store.init()
|
||||
|
||||
canned = [_canned_evidence(_transcript_for(sid)) for sid in _SCENARIO_IDS]
|
||||
llm = _ScriptedLLM(canned)
|
||||
|
||||
results: list[dict[str, Any]] = []
|
||||
for sid in _SCENARIO_IDS:
|
||||
rec = SessionRecorder(store, scenario_id=sid)
|
||||
await rec.start()
|
||||
rec.set_mastery_turns(_transcript_for(sid))
|
||||
rec.set_branch_path(["accept_resolution"])
|
||||
await rec.end(outcome="success", debrief_text="nicely done")
|
||||
res = await rec.run_mastery_flow(_deps(llm, sid))
|
||||
results.append(res)
|
||||
|
||||
# ── Check 1: every session scored (no scoring_inconclusive) ──
|
||||
for i, r in enumerate(results):
|
||||
if r["status"] != "scored":
|
||||
failures.append(
|
||||
f"session[{i}] ({_SCENARIO_IDS[i]}) status={r['status']!r} (expected 'scored')"
|
||||
)
|
||||
|
||||
# ── Check 2: every scenario passed ──
|
||||
for i, r in enumerate(results):
|
||||
if not r.get("passed"):
|
||||
failures.append(
|
||||
f"session[{i}] ({_SCENARIO_IDS[i]}) passed=False (mean={r.get('weighted_mean')})"
|
||||
)
|
||||
|
||||
# ── Check 3: theta converges upward (3 passes against increasing b) ──
|
||||
thetas = [r["theta"] for r in results]
|
||||
if not (thetas[-1] > DEFAULT_THETA and thetas[-1] >= thetas[0]):
|
||||
failures.append(
|
||||
f"theta did not converge upward: start={DEFAULT_THETA} "
|
||||
f"trajectory={thetas}"
|
||||
)
|
||||
|
||||
# ── Check 4: gate opens on the 3rd passing scenario ──
|
||||
gate_opens = [bool(r.get("gate_open")) for r in results]
|
||||
if not gate_opens[-1]:
|
||||
failures.append(
|
||||
f"gate did not open on 3rd passing scenario: gate_open={gate_opens}"
|
||||
)
|
||||
|
||||
# ── Check 5: gate-open path score >= 3.5 ──
|
||||
final_path_score = results[-1].get("weighted_mean", 0.0)
|
||||
progress_row = await store.get_progress(HARDCODED_LEARNER_ID, _PATH_SLUG)
|
||||
stored_score = float(progress_row["mastery_score"]) if progress_row else 0.0
|
||||
if stored_score < 3.5:
|
||||
failures.append(
|
||||
f"stored path mastery_score {stored_score} < 3.5 (gate threshold)"
|
||||
)
|
||||
|
||||
# ── Check 6: progress advanced at least once (new_week > 1 by end) ──
|
||||
if progress_row is None:
|
||||
failures.append("no mastery_progress row persisted")
|
||||
else:
|
||||
# After 3 passing scenarios the learner should have advanced weeks.
|
||||
if progress_row["current_week"] < 2:
|
||||
failures.append(
|
||||
f"progress did not advance: current_week={progress_row['current_week']}"
|
||||
)
|
||||
|
||||
# ── Check 7: gate events recorded (one per scored session) ──
|
||||
events = await store.list_gate_events(HARDCODED_LEARNER_ID, _PATH_SLUG)
|
||||
if len(events) != 3:
|
||||
failures.append(
|
||||
f"expected 3 gate events, got {len(events)}"
|
||||
)
|
||||
for ev in events:
|
||||
sp = json.loads(ev["scenarios_passed_json"])
|
||||
rs = json.loads(ev["rubric_scores_json"])
|
||||
if not isinstance(sp, list):
|
||||
failures.append(f"gate event {ev['id']} scenarios_passed_json not a list")
|
||||
if not isinstance(rs, list) or len(rs) != 4:
|
||||
failures.append(
|
||||
f"gate event {ev['id']} rubric_scores_json malformed (len={len(rs) if isinstance(rs, list) else 'NaN'})"
|
||||
)
|
||||
|
||||
# ── Summary ──
|
||||
print("")
|
||||
print(f" scenario trajectory : {_SCENARIO_IDS}")
|
||||
print(f" theta trajectory : {[round(t, 4) for t in thetas]}")
|
||||
print(f" gate-open trajectory: {gate_opens}")
|
||||
print(f" stored path score : {stored_score}")
|
||||
print(
|
||||
f" progress current_week: {progress_row['current_week'] if progress_row else 'N/A'}"
|
||||
)
|
||||
print(f" gate events recorded: {len(events)}")
|
||||
print("")
|
||||
|
||||
if failures:
|
||||
for f in failures:
|
||||
print(_fmt_fail("check", f))
|
||||
print("")
|
||||
print("RESULT: FAIL")
|
||||
return 1
|
||||
|
||||
checks = [
|
||||
"all 3 sessions scored",
|
||||
"all 3 scenarios passed",
|
||||
f"theta converged upward ({round(thetas[0], 3)} → {round(thetas[-1], 3)})",
|
||||
"gate opened on 3rd passing scenario",
|
||||
f"path score {stored_score} >= 3.5",
|
||||
"progress advanced week-by-week",
|
||||
"3 gate events recorded with parsable JSON evidence",
|
||||
]
|
||||
for c in checks:
|
||||
print(_fmt_pass(c))
|
||||
print("")
|
||||
print("RESULT: PASS")
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
return asyncio.run(_run())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env python3
|
||||
"""SLICE-08 TASK-08-04 — Real-LLM evidence extraction smoke test (grill Axis 7 FIX #1).
|
||||
|
||||
Runs ONE real session transcript through the actual deepseek-v4-flash:cloud
|
||||
evidence extractor and verifies the output is valid JSON with fuzzy-matching
|
||||
quotes (the extraction prompt works against the real model, not just the
|
||||
scoring logic against mocked responses).
|
||||
|
||||
Staging-gated: this test calls a real paid LLM endpoint. It runs ONLY when the
|
||||
env var `PRAXIS_RUN_REAL_LLM_TESTS=1` is set, AND requires `OLLAMA_API_KEY`.
|
||||
CI must NOT set the gate env var — mocked-LLM tests stay the CI source of
|
||||
truth (REQ-MAST-01 determinism is covered by the mocked tests; this script
|
||||
validates the prompt+model contract against model drift).
|
||||
|
||||
Run:
|
||||
python3 scripts/test_real_llm_evidence.py
|
||||
|
||||
Exit codes:
|
||||
0 — SKIP (gate not set) OR PASS
|
||||
1 — FAIL (gate set, real call failed or output invalid)
|
||||
2 — MISCONFIG (gate set but OLLAMA_API_KEY missing)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from server.llm.ollama_cloud import OllamaCloudLLM
|
||||
from server.mastery.evidence_extractor import extract_evidence
|
||||
from server.mastery.rubric_loader import clear_cache, load_rubric
|
||||
|
||||
_REPO = Path(__file__).resolve().parent.parent
|
||||
_RUBRICS_DIR = _REPO / "rubrics"
|
||||
_GATE_ENV = "PRAXIS_RUN_REAL_LLM_TESTS"
|
||||
|
||||
_TRANSCRIPT = [
|
||||
{"role": "customer", "content": "My order arrived cracked and I'm furious."},
|
||||
{
|
||||
"role": "learner",
|
||||
"content": (
|
||||
"I'm really sorry the bowl arrived cracked — that's genuinely "
|
||||
"frustrating. I can refund the full amount to your original card "
|
||||
"within 3 business days, or send a replacement first class tomorrow. "
|
||||
"Which would you prefer?"
|
||||
),
|
||||
},
|
||||
{"role": "customer", "content": "Just refund it."},
|
||||
{
|
||||
"role": "learner",
|
||||
"content": (
|
||||
"Of course — I've issued a full refund of $42.99 to your Visa ending "
|
||||
"4421. You'll see it in 2-3 business days. Is there anything else?"
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _print_skip() -> None:
|
||||
print(f"SKIP (set {_GATE_ENV}=1 to run)")
|
||||
|
||||
|
||||
async def _run_real() -> int:
|
||||
if not os.environ.get("OLLAMA_API_KEY", "").strip():
|
||||
print(f"FAIL — {_GATE_ENV}=1 but OLLAMA_API_KEY is not set")
|
||||
return 2
|
||||
|
||||
clear_cache()
|
||||
rubric = load_rubric("customer_service", rubrics_dir=_RUBRICS_DIR)
|
||||
llm = OllamaCloudLLM()
|
||||
|
||||
print("Calling deepseek-v4-flash:cloud for evidence extraction …")
|
||||
result = await extract_evidence(
|
||||
_TRANSCRIPT, rubric.criterion_ids(), llm, max_attempts=2
|
||||
)
|
||||
|
||||
if result.scoring_inconclusive:
|
||||
print(
|
||||
f"FAIL — extraction returned scoring_inconclusive after "
|
||||
f"{result.attempts} attempts; rejected quotes="
|
||||
f"{result.rejected_quotes[:3]}"
|
||||
)
|
||||
return 1
|
||||
|
||||
if not result.evidence:
|
||||
print(f"FAIL — extraction returned no evidence (attempts={result.attempts})")
|
||||
return 1
|
||||
|
||||
crit_ids = {e.criterion_id for e in result.evidence}
|
||||
expected = set(rubric.criterion_ids())
|
||||
if not crit_ids.issubset(expected):
|
||||
print(f"FAIL — unknown criterion ids: {crit_ids - expected}")
|
||||
return 1
|
||||
|
||||
for ev in result.evidence:
|
||||
if not ev.quote.strip():
|
||||
print(f"FAIL — empty quote for criterion {ev.criterion_id!r}")
|
||||
return 1
|
||||
if not ev.signals:
|
||||
print(f"FAIL — no signals for criterion {ev.criterion_id!r}")
|
||||
return 1
|
||||
|
||||
print(f"PASS — {len(result.evidence)} evidence items extracted (attempts={result.attempts})")
|
||||
for ev in result.evidence:
|
||||
print(f" - {ev.criterion_id}: {len(ev.signals)} signals, quote={ev.quote[:60]!r}…")
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
if os.environ.get(_GATE_ENV, "").strip() != "1":
|
||||
_print_skip()
|
||||
return 0
|
||||
return asyncio.run(_run_real())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user