"""SLICE-03 TASK-03-04 — scoring unit tests (mocked LLM).""" from __future__ import annotations import json from pathlib import Path from typing import Any from unittest.mock import AsyncMock import pytest from server.mastery.evidence_extractor import ( Evidence, ExtractionResult, extract_evidence, _fuzzy_contains, ) from server.mastery.mastery_score import ( check_gate, compute_path_score, compute_scenario_score, ) from server.mastery.rubric_loader import clear_cache, load_rubric from server.mastery.rubric_scorer import score _RUBRICS_DIR = Path(__file__).resolve().parent.parent / "rubrics" def _cs_rubric(): clear_cache() return load_rubric("customer_service", rubrics_dir=_RUBRICS_DIR) def _turns() -> list[dict]: return [ {"role": "customer", "content": "My order arrived cracked and I'm furious."}, { "role": "learner", "content": ( "I'm really sorry the bowl arrived cracked — that's genuinely " "frustrating. I can refund the full amount to your original card " "within 3 business days, or send a replacement first class tomorrow. " "Which would you prefer? I'll also log this so it doesn't happen again." ), }, {"role": "customer", "content": "Just refund it, this is ridiculous."}, { "role": "learner", "content": ( "Of course — I've issued a full refund of $42.99 to your Visa ending " "4421. You'll see it in 2-3 business days. Is there anything else I " "can help with today?" ), }, ] def _canned_evidence_json() -> str: learner_text = _turns()[1]["content"] learner_text2 = _turns()[3]["content"] return json.dumps( [ {"criterion_id": "empathy", "quote": learner_text, "signals": ["named_emotion_in_own_words", "acknowledged_specific"]}, {"criterion_id": "resolution", "quote": learner_text, "signals": ["concrete_method", "concrete_amount_or_channel", "concrete_next_step"]}, {"criterion_id": "de_escalation", "quote": learner_text, "signals": ["explicit_acknowledge_reframe_offer"]}, {"criterion_id": "professionalism", "quote": learner_text2, "signals": ["plain_language", "in_role_throughout", "no_prohibited_advice"]}, ] ) def _make_llm(raw_outputs: list[str]) -> AsyncMock: llm = AsyncMock() llm.chat_full = AsyncMock(side_effect=[(raw, {"model": "test"}) for raw in raw_outputs]) return llm # ── evidence extraction with mocked LLM ────────────────────────────────────── @pytest.mark.asyncio async def test_extract_evidence_happy_path(): rubric = _cs_rubric() llm = _make_llm([_canned_evidence_json()]) res = await extract_evidence(_turns(), rubric.criterion_ids(), llm) assert isinstance(res, ExtractionResult) assert not res.scoring_inconclusive assert res.attempts == 1 assert {e.criterion_id for e in res.evidence} == { "empathy", "resolution", "de_escalation", "professionalism", } for e in res.evidence: assert e.quote and e.signals @pytest.mark.asyncio async def test_extract_evidence_rejects_hallucinated_quote_then_recovers(): rubric = _cs_rubric() bad = json.dumps( [ {"criterion_id": "empathy", "quote": "I apologize for the inconvenience, customer.", "signals": ["named_emotion_in_own_words"]}, {"criterion_id": "resolution", "quote": "I can refund you.", "signals": ["concrete_method"]}, ] ) good = _canned_evidence_json() llm = _make_llm([bad, good]) res = await extract_evidence(_turns(), rubric.criterion_ids(), llm) assert not res.scoring_inconclusive assert res.attempts == 2 assert res.evidence assert any(e.criterion_id == "empathy" for e in res.evidence) @pytest.mark.asyncio async def test_extract_evidence_inconclusive_after_max_attempts(): rubric = _cs_rubric() bad = json.dumps( [{"criterion_id": "empathy", "quote": "totally invented text never spoken", "signals": ["named_emotion_in_own_words"]}] ) llm = _make_llm([bad, bad, bad]) res = await extract_evidence(_turns(), rubric.criterion_ids(), llm, max_attempts=2) assert res.scoring_inconclusive is True assert res.evidence == [] assert res.attempts == 3 def test_fuzzy_contains_exact_substring(): assert _fuzzy_contains("the quick brown fox", "quick brown") assert not _fuzzy_contains("the quick brown fox", "slow green") def test_fuzzy_contains_near_match_passes_at_threshold(): hay = "I'm really sorry the bowl arrived cracked — that's genuinely frustrating." quote = "I'm really sorry the bowl arrived cracked that's genuinely frustrating" # missing dash/period assert _fuzzy_contains(hay, quote) def test_fuzzy_contains_rejects_hallucination(): assert not _fuzzy_contains(_turns()[1]["content"], "I apologize for the inconvenience, customer.") # ── rule-based scoring determinism ──────────────────────────────────────────── def _make_evidence() -> list[Evidence]: return [ Evidence(criterion_id="empathy", quote="q1", signals=["named_emotion_in_own_words", "acknowledged_specific"]), Evidence(criterion_id="resolution", quote="q2", signals=["concrete_method", "concrete_amount_or_channel", "concrete_next_step"]), Evidence(criterion_id="de_escalation", quote="q3", signals=["explicit_acknowledge_reframe_offer"]), Evidence(criterion_id="professionalism", quote="q4", signals=["plain_language", "in_role_throughout", "no_prohibited_advice"]), ] def test_score_is_deterministic_same_output_twice(): rubric = _cs_rubric() ev = _make_evidence() a = score(ev, rubric) b = score(ev, rubric) assert [s.model_dump() for s in a] == [s.model_dump() for s in b] def test_score_maps_signals_to_highest_matching_level(): rubric = _cs_rubric() ev = _make_evidence() cs = {s.criterion_id: s for s in score(ev, rubric)} assert cs["empathy"].level == 3 assert cs["resolution"].level == 3 assert cs["de_escalation"].level == 3 assert cs["professionalism"].level == 3 def test_score_falls_back_to_level_1_on_no_evidence(): rubric = _cs_rubric() cs = {s.criterion_id: s for s in score([], rubric)} for s in cs.values(): assert s.level == 1 assert s.evidence_quote == "" def test_score_partial_signals_pick_lower_level(): rubric = _cs_rubric() ev = [Evidence(criterion_id="resolution", quote="q", signals=["concrete_method"])] cs = {s.criterion_id: s for s in score(ev, rubric)} assert cs["resolution"].level == 1 # ── conjunctive floor enforcement ───────────────────────────────────────────── def test_conjunctive_floor_fails_scenario_when_criterion_at_level_1(): rubric = _cs_rubric() ev = _make_evidence() ev = [e for e in ev if e.criterion_id != "professionalism"] ev.append(Evidence(criterion_id="professionalism", quote="x", signals=["unprofessional_language"])) all_scores = score(ev, rubric) prof = next(s for s in all_scores if s.criterion_id == "professionalism") assert prof.level == 1 ss = compute_scenario_score(all_scores, rubric) assert ss.passed is False assert "conjunctive_floor_violation:professionalism" in (ss.fail_reason or "") def test_conjunctive_floor_passes_when_all_criteria_above_floor(): rubric = _cs_rubric() ev = _make_evidence() all_scores = score(ev, rubric) assert all(s.level >= 2 for s in all_scores) ss = compute_scenario_score(all_scores, rubric) assert ss.passed is True assert ss.weighted_mean >= 3.0 def test_scenario_fails_when_mean_below_3_even_if_floors_ok(): rubric = _cs_rubric() ev = [ Evidence(criterion_id="empathy", quote="q1", signals=["scripted_empathy_line"]), Evidence(criterion_id="resolution", quote="q2", signals=["resolution_missing_specifics"]), Evidence(criterion_id="de_escalation", quote="q3", signals=["avoidance_or_deflection"]), Evidence(criterion_id="professionalism", quote="q4", signals=["uses_jargon", "breaks_tone_once"]), ] all_scores = score(ev, rubric) assert all(s.level >= 2 for s in all_scores) ss = compute_scenario_score(all_scores, rubric) assert ss.passed is False assert ss.fail_reason and "mean_below_threshold" in ss.fail_reason # ── gate logic ──────────────────────────────────────────────────────────────── def _ss(mean: float, passed: bool) -> Any: from server.mastery.mastery_score import ScenarioScore return ScenarioScore(criterion_scores=[], weighted_mean=mean, passed=passed, fail_reason=None if passed else "x") def test_gate_opens_at_3_passed_and_3_5(): path_score = compute_path_score([_ss(3.6, True), _ss(3.5, True), _ss(3.7, True)]) assert path_score >= 3.5 assert check_gate(path_score, 3) is True def test_gate_closes_with_only_2_passed(): path_score = compute_path_score([_ss(4.0, True), _ss(4.0, True)]) assert check_gate(path_score, 2) is False def test_gate_closes_at_3_passed_but_score_below_3_5(): path_score = compute_path_score([_ss(3.4, True), _ss(3.4, True), _ss(3.4, True)]) assert path_score < 3.5 assert check_gate(path_score, 3) is False def test_gate_opens_at_exactly_3_passed_and_3_5(): path_score = compute_path_score([_ss(3.5, True), _ss(3.5, True), _ss(3.5, True)]) assert path_score == 3.5 assert check_gate(path_score, 3) is True def test_path_score_ignores_failing_scenarios(): # compute_path_score is documented as "mean over passing scenarios only"; # the caller filters to passing before calling. path_score = compute_path_score([_ss(5.0, True), _ss(3.5, True), _ss(3.5, True)]) assert abs(path_score - 4.0) < 1e-6