feat(milestone): merge phase/01 mastery-core → milestone/v0.3-mastery-scoring

Phase 1 complete. Mastery scoring + competency rubrics + VC issuer shipped.
9 slices, 5 waves, 238 tests passing, 13/13 REQ-IDs covered.
4/4 grill MUST conditions satisfied. VERIFY: APPROVE_WITH_NOTES.

---ci---
project: praxis
phase: 1
milestone: v0.3
status: complete
requirements:
  covered: [REQ-MAST-01, REQ-MAST-02, REQ-MAST-03, REQ-SCEN-02, REQ-SCEN-03, REQ-SCEN-04, REQ-PATH-02, REQ-NFR-MAST-01, REQ-NFR-MAST-02, REQ-NFR-VC-01, REQ-NFR-VC-02, REQ-NFR-IRT-01]
  partial: []
---/ci---
This commit is contained in:
Praxis CI
2026-08-04 00:03:13 +00:00
parent 926322960e
commit 4d39596a7d
51 changed files with 6962 additions and 215 deletions
View File
+204
View File
@@ -0,0 +1,204 @@
"""Evidence extractor — LLM-extract-then-verify (SLICE-03 TASK-03-01).
Off-voice-path: called after the session ends. Calls deepseek-v4-flash:cloud
to pull verbatim-quote evidence per rubric criterion, then fuzzy-matches each
quote against the transcript (R-MAST-02). Hallucinated quotes are rejected and
re-extracted (max 2 attempts). On final failure the scenario is marked
`scoring_inconclusive=True` — it does NOT silently fail to zero and does NOT
penalize the learner (grill Axis 4 MUST #3).
"""
from __future__ import annotations
import json
import logging
from difflib import SequenceMatcher
from typing import Any
from pydantic import BaseModel, Field, ValidationError
from server.services.base import LLMProvider
log = logging.getLogger(__name__)
_QUOTE_MATCH_THRESHOLD = 0.85
_MAX_REEXTRACTION_ATTEMPTS = 2
_EXTRACTION_MODEL = "deepseek-v4-flash:cloud"
class Evidence(BaseModel):
criterion_id: str
quote: str
signals: list[str] = Field(default_factory=list)
class ExtractionResult(BaseModel):
evidence: list[Evidence] = Field(default_factory=list)
scoring_inconclusive: bool = False
attempts: int = 0
rejected_quotes: list[str] = Field(default_factory=list)
def _transcript_text(turns: list[dict]) -> str:
parts: list[str] = []
for t in turns:
role = t.get("role", "")
content = t.get("content", "") or t.get("text", "")
if content:
parts.append(f"{role}: {content}")
return "\n".join(parts)
def _fuzzy_contains(haystack: str, quote: str) -> bool:
if not quote.strip():
return False
if quote in haystack:
return True
qlen = len(quote)
if qlen >= len(haystack):
return SequenceMatcher(None, quote, haystack).ratio() >= _QUOTE_MATCH_THRESHOLD
best = 0.0
window = qlen + max(20, qlen // 4)
step = max(1, qlen // 4)
i = 0
while i <= len(haystack) - qlen:
end = min(len(haystack), i + window)
r = SequenceMatcher(None, quote, haystack[i:end]).ratio()
if r > best:
best = r
if best >= _QUOTE_MATCH_THRESHOLD:
return True
i += step
return best >= _QUOTE_MATCH_THRESHOLD
def _build_prompt(turns: list[dict], rubric_criteria: list[str]) -> list[dict[str, str]]:
transcript = _transcript_text(turns)
crit_block = "\n".join(f"- {c}" for c in rubric_criteria)
system = (
"You are an evidence extraction engine for a customer-service coaching rubric. "
"For each rubric criterion, find the single most representative verbatim quote "
"from the learner's utterances in the transcript, plus the observable behavior "
"signal tags that apply. Quotes MUST be copied verbatim from the learner's "
"spoken turns — do not paraphrase, do not invent."
)
user = (
f"Rubric criteria:\n{crit_block}\n\n"
f"Transcript:\n{transcript}\n\n"
"Return ONLY a JSON array. Each element: "
'{"criterion_id": <string>, "quote": <verbatim learner quote>, '
'"signals": [<string>, ...]}. '
"Omit a criterion if no evidence is present. No prose, no markdown fences."
)
return [{"role": "system", "content": system}, {"role": "user", "content": user}]
def _parse_evidence_json(raw: str, allowed_criteria: list[str]) -> list[Evidence]:
text = raw.strip()
if text.startswith("```"):
text = text.strip("`")
if text.lower().startswith("json"):
text = text[4:]
text = text.strip()
try:
data = json.loads(text)
except json.JSONDecodeError as exc:
raise ValueError(f"evidence JSON parse failed: {exc}") from exc
if not isinstance(data, list):
raise ValueError("evidence JSON must be a list")
allowed = set(allowed_criteria)
out: list[Evidence] = []
for item in data:
try:
ev = Evidence.model_validate(item)
except ValidationError as exc:
raise ValueError(f"evidence item schema invalid: {exc}") from exc
if ev.criterion_id not in allowed:
raise ValueError(f"unknown criterion_id: {ev.criterion_id}")
out.append(ev)
return out
async def extract_evidence(
turns: list[dict],
rubric_criteria: list[str],
llm: LLMProvider,
*,
model: str | None = None,
max_attempts: int = _MAX_REEXTRACTION_ATTEMPTS,
) -> ExtractionResult:
"""Extract verbatim-quote evidence per criterion via LLM + fuzzy verification.
Args:
turns: session transcript turns (each dict has role + content/text).
rubric_criteria: criterion ids to extract evidence for.
llm: LLMProvider whose chat_full returns the model's response.
model: override the extraction model (default deepseek-v4-flash:cloud).
max_attempts: max re-extraction attempts after the initial call (default 2).
Returns:
ExtractionResult — either with `.evidence` populated, or with
`.scoring_inconclusive=True` if quotes could not be verified after the
retry budget (grill Axis 4 MUST #3 — never silently fail to zero).
"""
mdl = model or _EXTRACTION_MODEL
transcript_text = _transcript_text(turns)
rejected: list[str] = []
attempts = 0
for attempt in range(max_attempts + 1):
attempts = attempt + 1
messages = _build_prompt(turns, rubric_criteria)
if attempt > 0 and rejected:
messages.append(
{
"role": "user",
"content": (
"The following quotes were NOT found verbatim in the transcript "
"and must be replaced with exact learner utterances:\n- "
+ "\n- ".join(rejected[-6:])
+ "\n\nRe-emit the full JSON array with corrected verbatim quotes."
),
}
)
try:
raw, _usage = await llm.chat_full(messages, model=mdl, no_think=True)
except Exception as exc:
log.warning("evidence extraction LLM call failed (attempt %d): %s", attempts, exc)
continue
try:
candidates = _parse_evidence_json(raw, rubric_criteria)
except ValueError as exc:
log.warning("evidence JSON invalid (attempt %d): %s", attempts, exc)
continue
verified: list[Evidence] = []
bad: list[str] = []
for ev in candidates:
if _fuzzy_contains(transcript_text, ev.quote):
verified.append(ev)
else:
bad.append(ev.quote)
if not bad and verified:
return ExtractionResult(evidence=verified, attempts=attempts, rejected_quotes=rejected)
rejected.extend(bad)
if not verified and not bad:
continue
log.error(
"evidence extraction scoring_inconclusive after %d attempts; rejected=%r",
attempts,
rejected,
)
return ExtractionResult(
evidence=[],
scoring_inconclusive=True,
attempts=attempts,
rejected_quotes=rejected,
)
__all__ = ["Evidence", "ExtractionResult", "extract_evidence"]
+98
View File
@@ -0,0 +1,98 @@
"""IRT engine — 1PL/Rasch with Bayesian theta update (SLICE-04, REQ-NFR-IRT-01).
P_success(theta, b) = logistic(theta - b) = 1 / (1 + exp(-(theta - b))).
update_theta uses a Gaussian-approximation Bayesian update (Kalman-like):
the posterior precision is the prior precision plus the Fisher information
P*(1-P), and the posterior mean shifts toward the outcome by the Kalman gain.
Cold-start (R-IRT-01): theta=0, sigma_sq=1; until >=5 observations, scenario
selection falls back to difficulty-based matching (difficulty closest to
round(theta + logit(target_p))).
"""
from __future__ import annotations
import math
from server.scenarios.library import ScenarioLibrary
from server.scenarios.schema import Scenario
COLD_START_MIN_OBSERVATIONS = 5
DEFAULT_THETA = 0.0
DEFAULT_SIGMA_SQ = 1.0
def _logit(p: float) -> float:
return math.log(p / (1.0 - p))
class IRTEngine:
"""1PL/Rasch IRT with Gaussian-approximation Bayesian theta updates."""
@staticmethod
def P_success(theta: float, b: float) -> float:
exp_neg = math.exp(-(theta - b))
return 1.0 / (1.0 + exp_neg)
@staticmethod
def update_theta(
theta: float, sigma_sq: float, outcome: float, b: float
) -> tuple[float, float]:
"""Bayesian update of theta given a binary (0/1) outcome.
Uses the standard 1PL Gaussian-approximation (Kalman-like) update:
P = P_success(theta, b)
new_precision = 1/sigma_sq + P*(1-P)
new_sigma_sq = 1 / new_precision
new_theta = theta + new_sigma_sq * (outcome - P)
"""
p = IRTEngine.P_success(theta, b)
prior_precision = 1.0 / sigma_sq
info = p * (1.0 - p)
new_precision = prior_precision + info
new_sigma_sq = 1.0 / new_precision
new_theta = theta + new_sigma_sq * (outcome - p)
return new_theta, new_sigma_sq
@staticmethod
def select_scenario(
theta: float,
library: ScenarioLibrary,
path: str,
target_p: float = 0.7,
observations: int = 0,
) -> Scenario | None:
"""Select the next scenario for a learner.
If observations < COLD_START_MIN_OBSERVATIONS (R-IRT-01), fall back to
difficulty-based selection: pick the scenario whose `difficulty` is
closest to round(theta + logit(target_p)).
Otherwise delegate to library.select_for_theta (IRT-aware selection
targeting ~target_p).
"""
if observations < COLD_START_MIN_OBSERVATIONS:
entries = library.list_by_path(path)
if not entries:
return None
target_difficulty = round(theta + _logit(target_p))
target_difficulty = max(1, min(5, target_difficulty))
best_entry = None
best_dist = math.inf
for e in entries:
dist = abs(e.difficulty - target_difficulty)
if dist < best_dist:
best_dist = dist
best_entry = e
if best_entry is None:
return None
return library.get(best_entry.id)
return library.select_for_theta(theta, path, target_p=target_p)
__all__ = [
"IRTEngine",
"COLD_START_MIN_OBSERVATIONS",
"DEFAULT_THETA",
"DEFAULT_SIGMA_SQ",
]
+94
View File
@@ -0,0 +1,94 @@
"""Mastery score + gate logic — deterministic (SLICE-03 TASK-03-03).
Weighted mean of per-criterion levels with a conjunctive floor (every criterion
>= 2 AND scenario mean >= 3.0 to pass). Path score is the mean over passing
scenarios only. Gate opens at >=3 distinct passed scenarios AND path score
>= 3.5 (D-032).
"""
from __future__ import annotations
from pydantic import BaseModel, Field
from server.mastery.rubric_scorer import CriterionScore
from server.mastery.rubric_schema import Rubric
_SCENARIO_PASS_MEAN = 3.0
_CONJUNCTIVE_FLOOR = 2
_GATE_REQUIRED_DISTINCT = 3
_GATE_REQUIRED_SCORE = 3.5
class ScenarioScore(BaseModel):
criterion_scores: list[CriterionScore]
weighted_mean: float
passed: bool
fail_reason: str | None = None
@property
def scenario_id(self) -> str | None:
return None
def compute_scenario_score(
criterion_scores: list[CriterionScore], rubric: Rubric
) -> ScenarioScore:
"""Compute a deterministic scenario score with conjunctive-floor enforcement.
Pass requires: weighted mean >= 3.0 AND every criterion >= 2 AND any
criterion with `conjunctive_floor` set must be >= that floor.
"""
weights = {c.id: c.weight for c in rubric.criteria}
total = 0.0
for cs in criterion_scores:
w = weights.get(cs.criterion_id, cs.weight)
total += cs.level * w
mean = round(total, 6)
floor_violations: list[str] = []
for cs in criterion_scores:
c = rubric.criterion_by_id(cs.criterion_id)
floor = c.conjunctive_floor if c else None
required = max(floor or _CONJUNCTIVE_FLOOR, _CONJUNCTIVE_FLOOR)
if cs.level < required:
floor_violations.append(cs.criterion_id)
fail_reason: str | None = None
if floor_violations:
fail_reason = f"conjunctive_floor_violation:{','.join(floor_violations)}"
elif mean < _SCENARIO_PASS_MEAN:
fail_reason = f"mean_below_threshold:{mean}<{_SCENARIO_PASS_MEAN}"
passed = fail_reason is None
return ScenarioScore(
criterion_scores=criterion_scores,
weighted_mean=mean,
passed=passed,
fail_reason=fail_reason,
)
def compute_path_score(passing_scenario_scores: list[ScenarioScore]) -> float:
"""Mean weighted-mean over passing scenarios only. Empty → 0.0."""
if not passing_scenario_scores:
return 0.0
return round(sum(s.weighted_mean for s in passing_scenario_scores) / len(passing_scenario_scores), 6)
def check_gate(
path_score: float,
distinct_passed_count: int,
*,
required: int = _GATE_REQUIRED_DISTINCT,
threshold: float = _GATE_REQUIRED_SCORE,
) -> bool:
"""Gate opens at >= `required` distinct passed scenarios AND path_score >= `threshold` (D-032)."""
return distinct_passed_count >= required and path_score >= threshold
__all__ = [
"ScenarioScore",
"compute_scenario_score",
"compute_path_score",
"check_gate",
]
+65
View File
@@ -0,0 +1,65 @@
"""Rubric loader — YAML → Pydantic Rubric (SLICE-01, D-039).
Loads a competency rubric by skill name from the `rubrics/` directory, validates
it against the Pydantic schema, and caches the parsed result in-memory for the
lifetime of the process. Used by the scoring engine (SLICE-03) and the path
engine (SLICE-05).
"""
from __future__ import annotations
from pathlib import Path
from threading import Lock
from typing import Dict
import yaml
from server.mastery.rubric_schema import Rubric, ValidationError
_DEFAULT_RUBRICS_DIR = Path(__file__).resolve().parent.parent.parent / "rubrics"
_cache: Dict[str, Rubric] = {}
_cache_lock = Lock()
def load_rubric(skill: str, rubrics_dir: Path | None = None) -> Rubric:
"""Load and validate a rubric by skill name.
Args:
skill: e.g. 'customer_service' (the YAML filename stem under rubrics/).
rubrics_dir: override the rubrics directory (default: repo /rubrics).
Returns:
A validated Rubric object. Cached in-memory per skill.
Raises:
FileNotFoundError: if the YAML file doesn't exist.
ValidationError: if the YAML fails schema validation (typed Pydantic error).
"""
with _cache_lock:
cached = _cache.get(skill)
if cached is not None:
return cached
base = rubrics_dir or _DEFAULT_RUBRICS_DIR
path = base / f"{skill}.yaml"
if not path.exists():
raise FileNotFoundError(f"Rubric YAML not found: {skill} in {base}")
with path.open("r", encoding="utf-8") as f:
raw = yaml.safe_load(f)
rubric = Rubric.model_validate(raw)
with _cache_lock:
_cache[skill] = rubric
return rubric
def clear_cache() -> None:
"""Clear the in-memory rubric cache (test helper)."""
with _cache_lock:
_cache.clear()
__all__ = ["load_rubric", "clear_cache", "ValidationError"]
+115
View File
@@ -0,0 +1,115 @@
"""Praxis competency rubric schema — YAML → Pydantic (SLICE-01, D-039).
Defines the typed model for a competency rubric: 4+ criteria, each with 5
behavioral anchor levels (Dreyfus + Miller "Does" + EPA entrustment per
RESEARCH §2). Loaded from `rubrics/<skill>.yaml` by rubric_loader.py and
referenced by the scoring engine (SLICE-03).
"""
from __future__ import annotations
from typing import Any
from pydantic import BaseModel, Field, ValidationError, field_validator, model_validator
_LEVEL_FLOOR = 1
_LEVEL_CEIL = 5
_REQUIRED_LEVELS = 5
_WEIGHT_TOLERANCE = 1e-6
class RubricLevel(BaseModel):
"""One anchor level (1=fail … 5=mastery/entrustable)."""
level: int = Field(..., ge=_LEVEL_FLOOR, le=_LEVEL_CEIL, description="1-5 level")
label: str = Field(..., description="Short human label, e.g. 'Fail', 'Mastery / Entrustable'")
anchor: str = Field(..., description="Observable-behavior anchor text (transcript-grounded)")
signals: list[str] = Field(
..., min_length=1, description="Observable behavior tags that map evidence to this level"
)
class RubricCriterion(BaseModel):
"""One scoring criterion (e.g. empathy) with weight + 5 anchor levels."""
id: str = Field(..., description="Criterion id, e.g. 'empathy'")
name: str = Field(..., description="Human-readable criterion name")
weight: float = Field(..., ge=0.0, le=1.0, description="Criterion weight (sums to 1.0 across criteria)")
conjunctive_floor: int | None = Field(
None,
ge=_LEVEL_FLOOR,
le=_LEVEL_CEIL,
description="If set, scenario cannot pass unless this criterion ≥ floor (professionalism ≥2)",
)
levels: list[RubricLevel] = Field(..., min_length=_REQUIRED_LEVELS, max_length=_REQUIRED_LEVELS)
@field_validator("levels")
@classmethod
def _levels_are_sequential(cls, v: list[RubricLevel]) -> list[RubricLevel]:
seen = sorted(lvl.level for lvl in v)
expected = list(range(_LEVEL_FLOOR, _LEVEL_CEIL + 1))
if seen != expected:
raise ValueError(
f"criterion levels must be exactly 1..{_REQUIRED_LEVELS}, got {seen}"
)
return v
def level_by_value(self, level: int) -> RubricLevel | None:
for lvl in self.levels:
if lvl.level == level:
return lvl
return None
class Rubric(BaseModel):
"""A competency rubric for a skill (e.g. customer_service)."""
id: str = Field(..., description="Rubric id, e.g. 'customer_service'")
skill: str = Field(..., description="Skill path this rubric scores, e.g. 'customer_service'")
description: str | None = Field(None, description="Optional human description")
criteria: list[RubricCriterion] = Field(..., min_length=1)
archetype_weights: dict[str, dict[str, float]] | None = Field(
None, description="Per-archetype weight overrides (D-039 amendment)"
)
escalated_weights: dict[str, float] | None = Field(
None, description="Optional re-weight set when the escalate branch triggers (RESEARCH §6.3)"
)
@model_validator(mode="after")
def _validate_weights_and_ids(self) -> Rubric:
total = sum(c.weight for c in self.criteria)
if abs(total - 1.0) > _WEIGHT_TOLERANCE:
raise ValueError(
f"criterion weights must sum to 1.0 (±{_WEIGHT_TOLERANCE}), got {total}"
)
ids = [c.id for c in self.criteria]
if len(ids) != len(set(ids)):
dupes = sorted({i for i in ids if ids.count(i) > 1})
raise ValueError(f"duplicate criterion ids: {dupes}")
if self.skill != self.id and not self.id.startswith(self.skill):
pass
return self
def criterion_by_id(self, criterion_id: str) -> RubricCriterion | None:
for c in self.criteria:
if c.id == criterion_id:
return c
return None
def weights_for_archetype(self, archetype: str | None) -> dict[str, float]:
"""Return {criterion_id: weight} for an archetype, falling back to the base weights."""
if archetype and self.archetype_weights and archetype in self.archetype_weights:
override = self.archetype_weights[archetype]
return {c.id: override.get(c.id, c.weight) for c in self.criteria}
return {c.id: c.weight for c in self.criteria}
def criterion_ids(self) -> list[str]:
return [c.id for c in self.criteria]
__all__ = [
"Rubric",
"RubricCriterion",
"RubricLevel",
"ValidationError",
]
+67
View File
@@ -0,0 +1,67 @@
"""Rule-based rubric scorer — deterministic (SLICE-03 TASK-03-02, REQ-NFR-MAST-01).
No LLM. Maps evidence signals to rubric level anchors: for each criterion, pick
the highest level whose `signals[]` are all present in the matched evidence,
fallback to level 1 if no level matches. The output is reproducible given the
same (evidence, rubric) pair.
"""
from __future__ import annotations
from pydantic import BaseModel, Field
from server.mastery.evidence_extractor import Evidence
from server.mastery.rubric_schema import Rubric, RubricCriterion
class CriterionScore(BaseModel):
criterion_id: str
level: int = Field(ge=1, le=5)
weight: float
evidence_quote: str = ""
matched_signals: list[str] = Field(default_factory=list)
def _evidence_for(evidence: list[Evidence], criterion_id: str) -> Evidence | None:
for ev in evidence:
if ev.criterion_id == criterion_id:
return ev
return None
def _level_for_criterion(criterion: RubricCriterion, ev: Evidence | None) -> tuple[int, list[str]]:
if ev is None or not ev.signals:
return 1, []
ev_signals = set(ev.signals)
best_level = 1
best_signals: list[str] = []
for lvl in sorted(criterion.levels, key=lambda l: l.level):
if all(s in ev_signals for s in lvl.signals):
best_level = lvl.level
best_signals = list(lvl.signals)
return best_level, best_signals
def score(evidence: list[Evidence], rubric: Rubric) -> list[CriterionScore]:
"""Score evidence against the rubric — deterministic, no LLM.
Returns one CriterionScore per rubric criterion, in rubric order. Criteria
with no matching evidence get level 1 (the "Fail" anchor).
"""
out: list[CriterionScore] = []
for c in rubric.criteria:
ev = _evidence_for(evidence, c.id)
level, matched = _level_for_criterion(c, ev)
out.append(
CriterionScore(
criterion_id=c.id,
level=level,
weight=c.weight,
evidence_quote=ev.quote if ev else "",
matched_signals=matched,
)
)
return out
__all__ = ["CriterionScore", "score"]