Files
praxis/scripts/test_mastery_e2e.py
T
Praxis CI 813bd586d6 docs(milestone): merge v0.3-mastery-scoring → main
v0.3 milestone merged to main. Mastery scoring + competency rubrics +
verifiable credentials (formative-tier) shipped. 13/13 REQ-IDs covered.
Next milestone: v0.4 (operator tier — cohort dashboard + auth + Postgres).

---ci---
project: praxis
phase: 2
milestone: v0.3
status: complete
milestone_complete: true
milestone_merged_to_main: true
---/ci---
2026-08-04 00:14:59 +00:00

316 lines
12 KiB
Python

#!/usr/bin/env python3
"""SLICE-08 TASK-08-01 — End-to-end P1 mastery smoke test (not a pytest).
Simulates 3 sessions across 3 distinct Customer-Service scenarios → runs the
mastery flow (with a mocked LLM returning canned verbatim-quote evidence) →
verifies:
- mastery gate opens after the 3rd passing scenario with path score >= 3.5
- theta converges upward (passes against increasing difficulty)
- progress advances week-by-week as each week's gate opens
- one mastery_gate_event row is recorded per session in SQLite
Runnable: `python3 scripts/test_mastery_e2e.py`
Exit code 0 on PASS, 1 on FAIL. Prints a PASS/FAIL summary.
"""
from __future__ import annotations
import asyncio
import json
import os
import sys
import tempfile
from pathlib import Path
from typing import Any
from unittest.mock import AsyncMock
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from db.store import PraxisStore, HARDCODED_LEARNER_ID
from server.mastery.irt import IRTEngine, DEFAULT_THETA
from server.mastery.rubric_loader import clear_cache, load_rubric
from server.paths.engine import PathEngine
from server.scenarios.loader import load as load_scenario
from server.session_recorder import MasteryFlowDeps, SessionRecorder
_REPO = Path(__file__).resolve().parent.parent
_RUBRICS_DIR = _REPO / "rubrics"
_SCENARIOS_DIR = _REPO / "scenarios"
_PATHS_DIR = _REPO / "paths"
_PATH_SLUG = "customer_service"
_SCENARIO_IDS = [
"cs_refund_ca_v01",
"cs_escalation_ca_v02",
"cs_policy_exception_ca_v03",
]
def _transcript_for(scenario_id: str) -> list[dict[str, str]]:
if scenario_id == "cs_refund_ca_v01":
learner_a = (
"I'm really sorry the bowl arrived cracked — that's genuinely "
"frustrating. I can refund the full amount to your original card "
"within 3 business days, or send a replacement first class tomorrow. "
"Which would you prefer?"
)
learner_b = (
"Of course — I've issued a full refund of $42.99 to your Visa ending "
"4421. You'll see it in 2-3 business days. Is there anything else I "
"can help with today?"
)
elif scenario_id == "cs_escalation_ca_v02":
learner_a = (
"I hear you — two weeks with no straight answers is genuinely "
"infuriating, and you're right to push for clarity. I'm not going to "
"hide behind policy. Here's what I can do right now: I'll trace the "
"shipment, refund the shipping cost today, and give you a firm "
"delivery date within 24 hours. Would that work?"
)
learner_b = (
"Thank you for staying with me on this. I've refunded the $9.50 "
"shipping charge to your card and flagged the order for immediate "
"dispatch. You'll get a tracking number by email within the hour. "
"Is there anything else I can do for you?"
)
else:
learner_a = (
"You're absolutely right — a defect appearing last week is a "
"different situation from a 45-day change-of-mind. The 30-day window "
"is a guideline for returns, not a hard wall for defects. I can "
"offer a partial credit of 70% toward a replacement, or start a "
"manufacturer warranty claim on your behalf. Which would you prefer?"
)
learner_b = (
"I've issued a $30 partial credit to your original payment method "
"and started the manufacturer warranty claim — they'll reach out "
"within 5 business days. You'll get a confirmation email within the "
"hour. Anything else I can help with today?"
)
return [
{"role": "customer", "content": "I'm upset and need this resolved now."},
{"role": "learner", "content": learner_a},
{"role": "customer", "content": "Okay, go ahead with that."},
{"role": "learner", "content": learner_b},
]
def _canned_evidence(transcript: list[dict[str, str]]) -> str:
t1 = transcript[1]["content"]
t2 = transcript[3]["content"]
return json.dumps(
[
{
"criterion_id": "empathy",
"quote": t1,
"signals": [
"named_emotion_in_own_words",
"acknowledged_specific",
"tone_pace_adjusted",
"multiple_acknowledgement_instances",
],
},
{
"criterion_id": "resolution",
"quote": t1,
"signals": [
"concrete_method",
"concrete_amount_or_channel",
"concrete_next_step",
"decision_tree_of_options",
"matched_to_customer_preference",
"confirms_acceptance",
],
},
{
"criterion_id": "de_escalation",
"quote": t1,
"signals": [
"explicit_acknowledge_reframe_offer",
"cycles_acknowledge_reframe",
"lowers_intensity_without_conceding_policy",
],
},
{
"criterion_id": "professionalism",
"quote": t2,
"signals": [
"plain_language",
"in_role_throughout",
"no_prohibited_advice",
"adapts_register",
"concise_for_voice",
"manages_silence",
],
},
]
)
class _ScriptedLLM:
def __init__(self, raws: list[str]) -> None:
self._iter = iter(raws)
async def chat_full(
self,
messages: list[dict[str, str]],
*,
model: str | None = None,
no_think: bool = False,
) -> tuple[str, dict[str, Any]]:
try:
raw = next(self._iter)
except StopIteration as exc:
raise RuntimeError("scripted LLM exhausted") from exc
return raw, {"model": model or "test"}
def _deps(llm: Any, scenario_id: str) -> MasteryFlowDeps:
clear_cache()
return MasteryFlowDeps(
llm=llm,
irt=IRTEngine(),
path_engine=PathEngine(paths_dir=_PATHS_DIR),
load_rubric=lambda: load_rubric(_PATH_SLUG, rubrics_dir=_RUBRICS_DIR),
load_scenario=lambda: load_scenario(scenario_id, scenarios_dir=_SCENARIOS_DIR),
load_path=lambda: PathEngine(paths_dir=_PATHS_DIR).load_path(_PATH_SLUG),
)
def _fmt_pass(label: str) -> str:
return f" PASS {label}"
def _fmt_fail(label: str, detail: str) -> str:
return f" FAIL {label}{detail}"
async def _run() -> int:
failures: list[str] = []
print("=" * 70)
print("SLICE-08 TASK-08-01 — End-to-end P1 mastery smoke test")
print("=" * 70)
with tempfile.TemporaryDirectory(prefix="praxis_e2e_") as tmp:
db_path = Path(tmp) / "e2e.db"
store = PraxisStore(db_path)
await store.init()
canned = [_canned_evidence(_transcript_for(sid)) for sid in _SCENARIO_IDS]
llm = _ScriptedLLM(canned)
results: list[dict[str, Any]] = []
for sid in _SCENARIO_IDS:
rec = SessionRecorder(store, scenario_id=sid)
await rec.start()
rec.set_mastery_turns(_transcript_for(sid))
rec.set_branch_path(["accept_resolution"])
await rec.end(outcome="success", debrief_text="nicely done")
res = await rec.run_mastery_flow(_deps(llm, sid))
results.append(res)
# ── Check 1: every session scored (no scoring_inconclusive) ──
for i, r in enumerate(results):
if r["status"] != "scored":
failures.append(
f"session[{i}] ({_SCENARIO_IDS[i]}) status={r['status']!r} (expected 'scored')"
)
# ── Check 2: every scenario passed ──
for i, r in enumerate(results):
if not r.get("passed"):
failures.append(
f"session[{i}] ({_SCENARIO_IDS[i]}) passed=False (mean={r.get('weighted_mean')})"
)
# ── Check 3: theta converges upward (3 passes against increasing b) ──
thetas = [r["theta"] for r in results]
if not (thetas[-1] > DEFAULT_THETA and thetas[-1] >= thetas[0]):
failures.append(
f"theta did not converge upward: start={DEFAULT_THETA} "
f"trajectory={thetas}"
)
# ── Check 4: gate opens on the 3rd passing scenario ──
gate_opens = [bool(r.get("gate_open")) for r in results]
if not gate_opens[-1]:
failures.append(
f"gate did not open on 3rd passing scenario: gate_open={gate_opens}"
)
# ── Check 5: gate-open path score >= 3.5 ──
final_path_score = results[-1].get("weighted_mean", 0.0)
progress_row = await store.get_progress(HARDCODED_LEARNER_ID, _PATH_SLUG)
stored_score = float(progress_row["mastery_score"]) if progress_row else 0.0
if stored_score < 3.5:
failures.append(
f"stored path mastery_score {stored_score} < 3.5 (gate threshold)"
)
# ── Check 6: progress advanced at least once (new_week > 1 by end) ──
if progress_row is None:
failures.append("no mastery_progress row persisted")
else:
# After 3 passing scenarios the learner should have advanced weeks.
if progress_row["current_week"] < 2:
failures.append(
f"progress did not advance: current_week={progress_row['current_week']}"
)
# ── Check 7: gate events recorded (one per scored session) ──
events = await store.list_gate_events(HARDCODED_LEARNER_ID, _PATH_SLUG)
if len(events) != 3:
failures.append(
f"expected 3 gate events, got {len(events)}"
)
for ev in events:
sp = json.loads(ev["scenarios_passed_json"])
rs = json.loads(ev["rubric_scores_json"])
if not isinstance(sp, list):
failures.append(f"gate event {ev['id']} scenarios_passed_json not a list")
if not isinstance(rs, list) or len(rs) != 4:
failures.append(
f"gate event {ev['id']} rubric_scores_json malformed (len={len(rs) if isinstance(rs, list) else 'NaN'})"
)
# ── Summary ──
print("")
print(f" scenario trajectory : {_SCENARIO_IDS}")
print(f" theta trajectory : {[round(t, 4) for t in thetas]}")
print(f" gate-open trajectory: {gate_opens}")
print(f" stored path score : {stored_score}")
print(
f" progress current_week: {progress_row['current_week'] if progress_row else 'N/A'}"
)
print(f" gate events recorded: {len(events)}")
print("")
if failures:
for f in failures:
print(_fmt_fail("check", f))
print("")
print("RESULT: FAIL")
return 1
checks = [
"all 3 sessions scored",
"all 3 scenarios passed",
f"theta converged upward ({round(thetas[0], 3)}{round(thetas[-1], 3)})",
"gate opened on 3rd passing scenario",
f"path score {stored_score} >= 3.5",
"progress advanced week-by-week",
"3 gate events recorded with parsable JSON evidence",
]
for c in checks:
print(_fmt_pass(c))
print("")
print("RESULT: PASS")
return 0
def main() -> int:
return asyncio.run(_run())
if __name__ == "__main__":
raise SystemExit(main())