feat(milestone): merge phase/01 mastery-core → milestone/v0.3-mastery-scoring
Phase 1 complete. Mastery scoring + competency rubrics + VC issuer shipped. 9 slices, 5 waves, 238 tests passing, 13/13 REQ-IDs covered. 4/4 grill MUST conditions satisfied. VERIFY: APPROVE_WITH_NOTES. ---ci--- project: praxis phase: 1 milestone: v0.3 status: complete requirements: covered: [REQ-MAST-01, REQ-MAST-02, REQ-MAST-03, REQ-SCEN-02, REQ-SCEN-03, REQ-SCEN-04, REQ-PATH-02, REQ-NFR-MAST-01, REQ-NFR-MAST-02, REQ-NFR-VC-01, REQ-NFR-VC-02, REQ-NFR-IRT-01] partial: [] ---/ci---
This commit is contained in:
@@ -0,0 +1,204 @@
|
||||
"""Evidence extractor — LLM-extract-then-verify (SLICE-03 TASK-03-01).
|
||||
|
||||
Off-voice-path: called after the session ends. Calls deepseek-v4-flash:cloud
|
||||
to pull verbatim-quote evidence per rubric criterion, then fuzzy-matches each
|
||||
quote against the transcript (R-MAST-02). Hallucinated quotes are rejected and
|
||||
re-extracted (max 2 attempts). On final failure the scenario is marked
|
||||
`scoring_inconclusive=True` — it does NOT silently fail to zero and does NOT
|
||||
penalize the learner (grill Axis 4 MUST #3).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from difflib import SequenceMatcher
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field, ValidationError
|
||||
|
||||
from server.services.base import LLMProvider
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_QUOTE_MATCH_THRESHOLD = 0.85
|
||||
_MAX_REEXTRACTION_ATTEMPTS = 2
|
||||
_EXTRACTION_MODEL = "deepseek-v4-flash:cloud"
|
||||
|
||||
|
||||
class Evidence(BaseModel):
|
||||
criterion_id: str
|
||||
quote: str
|
||||
signals: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ExtractionResult(BaseModel):
|
||||
evidence: list[Evidence] = Field(default_factory=list)
|
||||
scoring_inconclusive: bool = False
|
||||
attempts: int = 0
|
||||
rejected_quotes: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
def _transcript_text(turns: list[dict]) -> str:
|
||||
parts: list[str] = []
|
||||
for t in turns:
|
||||
role = t.get("role", "")
|
||||
content = t.get("content", "") or t.get("text", "")
|
||||
if content:
|
||||
parts.append(f"{role}: {content}")
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _fuzzy_contains(haystack: str, quote: str) -> bool:
|
||||
if not quote.strip():
|
||||
return False
|
||||
if quote in haystack:
|
||||
return True
|
||||
qlen = len(quote)
|
||||
if qlen >= len(haystack):
|
||||
return SequenceMatcher(None, quote, haystack).ratio() >= _QUOTE_MATCH_THRESHOLD
|
||||
best = 0.0
|
||||
window = qlen + max(20, qlen // 4)
|
||||
step = max(1, qlen // 4)
|
||||
i = 0
|
||||
while i <= len(haystack) - qlen:
|
||||
end = min(len(haystack), i + window)
|
||||
r = SequenceMatcher(None, quote, haystack[i:end]).ratio()
|
||||
if r > best:
|
||||
best = r
|
||||
if best >= _QUOTE_MATCH_THRESHOLD:
|
||||
return True
|
||||
i += step
|
||||
return best >= _QUOTE_MATCH_THRESHOLD
|
||||
|
||||
|
||||
def _build_prompt(turns: list[dict], rubric_criteria: list[str]) -> list[dict[str, str]]:
|
||||
transcript = _transcript_text(turns)
|
||||
crit_block = "\n".join(f"- {c}" for c in rubric_criteria)
|
||||
system = (
|
||||
"You are an evidence extraction engine for a customer-service coaching rubric. "
|
||||
"For each rubric criterion, find the single most representative verbatim quote "
|
||||
"from the learner's utterances in the transcript, plus the observable behavior "
|
||||
"signal tags that apply. Quotes MUST be copied verbatim from the learner's "
|
||||
"spoken turns — do not paraphrase, do not invent."
|
||||
)
|
||||
user = (
|
||||
f"Rubric criteria:\n{crit_block}\n\n"
|
||||
f"Transcript:\n{transcript}\n\n"
|
||||
"Return ONLY a JSON array. Each element: "
|
||||
'{"criterion_id": <string>, "quote": <verbatim learner quote>, '
|
||||
'"signals": [<string>, ...]}. '
|
||||
"Omit a criterion if no evidence is present. No prose, no markdown fences."
|
||||
)
|
||||
return [{"role": "system", "content": system}, {"role": "user", "content": user}]
|
||||
|
||||
|
||||
def _parse_evidence_json(raw: str, allowed_criteria: list[str]) -> list[Evidence]:
|
||||
text = raw.strip()
|
||||
if text.startswith("```"):
|
||||
text = text.strip("`")
|
||||
if text.lower().startswith("json"):
|
||||
text = text[4:]
|
||||
text = text.strip()
|
||||
try:
|
||||
data = json.loads(text)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"evidence JSON parse failed: {exc}") from exc
|
||||
if not isinstance(data, list):
|
||||
raise ValueError("evidence JSON must be a list")
|
||||
allowed = set(allowed_criteria)
|
||||
out: list[Evidence] = []
|
||||
for item in data:
|
||||
try:
|
||||
ev = Evidence.model_validate(item)
|
||||
except ValidationError as exc:
|
||||
raise ValueError(f"evidence item schema invalid: {exc}") from exc
|
||||
if ev.criterion_id not in allowed:
|
||||
raise ValueError(f"unknown criterion_id: {ev.criterion_id}")
|
||||
out.append(ev)
|
||||
return out
|
||||
|
||||
|
||||
async def extract_evidence(
|
||||
turns: list[dict],
|
||||
rubric_criteria: list[str],
|
||||
llm: LLMProvider,
|
||||
*,
|
||||
model: str | None = None,
|
||||
max_attempts: int = _MAX_REEXTRACTION_ATTEMPTS,
|
||||
) -> ExtractionResult:
|
||||
"""Extract verbatim-quote evidence per criterion via LLM + fuzzy verification.
|
||||
|
||||
Args:
|
||||
turns: session transcript turns (each dict has role + content/text).
|
||||
rubric_criteria: criterion ids to extract evidence for.
|
||||
llm: LLMProvider whose chat_full returns the model's response.
|
||||
model: override the extraction model (default deepseek-v4-flash:cloud).
|
||||
max_attempts: max re-extraction attempts after the initial call (default 2).
|
||||
|
||||
Returns:
|
||||
ExtractionResult — either with `.evidence` populated, or with
|
||||
`.scoring_inconclusive=True` if quotes could not be verified after the
|
||||
retry budget (grill Axis 4 MUST #3 — never silently fail to zero).
|
||||
"""
|
||||
mdl = model or _EXTRACTION_MODEL
|
||||
transcript_text = _transcript_text(turns)
|
||||
rejected: list[str] = []
|
||||
attempts = 0
|
||||
|
||||
for attempt in range(max_attempts + 1):
|
||||
attempts = attempt + 1
|
||||
messages = _build_prompt(turns, rubric_criteria)
|
||||
if attempt > 0 and rejected:
|
||||
messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
"The following quotes were NOT found verbatim in the transcript "
|
||||
"and must be replaced with exact learner utterances:\n- "
|
||||
+ "\n- ".join(rejected[-6:])
|
||||
+ "\n\nRe-emit the full JSON array with corrected verbatim quotes."
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
try:
|
||||
raw, _usage = await llm.chat_full(messages, model=mdl, no_think=True)
|
||||
except Exception as exc:
|
||||
log.warning("evidence extraction LLM call failed (attempt %d): %s", attempts, exc)
|
||||
continue
|
||||
|
||||
try:
|
||||
candidates = _parse_evidence_json(raw, rubric_criteria)
|
||||
except ValueError as exc:
|
||||
log.warning("evidence JSON invalid (attempt %d): %s", attempts, exc)
|
||||
continue
|
||||
|
||||
verified: list[Evidence] = []
|
||||
bad: list[str] = []
|
||||
for ev in candidates:
|
||||
if _fuzzy_contains(transcript_text, ev.quote):
|
||||
verified.append(ev)
|
||||
else:
|
||||
bad.append(ev.quote)
|
||||
|
||||
if not bad and verified:
|
||||
return ExtractionResult(evidence=verified, attempts=attempts, rejected_quotes=rejected)
|
||||
rejected.extend(bad)
|
||||
if not verified and not bad:
|
||||
continue
|
||||
|
||||
log.error(
|
||||
"evidence extraction scoring_inconclusive after %d attempts; rejected=%r",
|
||||
attempts,
|
||||
rejected,
|
||||
)
|
||||
return ExtractionResult(
|
||||
evidence=[],
|
||||
scoring_inconclusive=True,
|
||||
attempts=attempts,
|
||||
rejected_quotes=rejected,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["Evidence", "ExtractionResult", "extract_evidence"]
|
||||
@@ -0,0 +1,98 @@
|
||||
"""IRT engine — 1PL/Rasch with Bayesian theta update (SLICE-04, REQ-NFR-IRT-01).
|
||||
|
||||
P_success(theta, b) = logistic(theta - b) = 1 / (1 + exp(-(theta - b))).
|
||||
update_theta uses a Gaussian-approximation Bayesian update (Kalman-like):
|
||||
the posterior precision is the prior precision plus the Fisher information
|
||||
P*(1-P), and the posterior mean shifts toward the outcome by the Kalman gain.
|
||||
|
||||
Cold-start (R-IRT-01): theta=0, sigma_sq=1; until >=5 observations, scenario
|
||||
selection falls back to difficulty-based matching (difficulty closest to
|
||||
round(theta + logit(target_p))).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
|
||||
from server.scenarios.library import ScenarioLibrary
|
||||
from server.scenarios.schema import Scenario
|
||||
|
||||
COLD_START_MIN_OBSERVATIONS = 5
|
||||
DEFAULT_THETA = 0.0
|
||||
DEFAULT_SIGMA_SQ = 1.0
|
||||
|
||||
|
||||
def _logit(p: float) -> float:
|
||||
return math.log(p / (1.0 - p))
|
||||
|
||||
|
||||
class IRTEngine:
|
||||
"""1PL/Rasch IRT with Gaussian-approximation Bayesian theta updates."""
|
||||
|
||||
@staticmethod
|
||||
def P_success(theta: float, b: float) -> float:
|
||||
exp_neg = math.exp(-(theta - b))
|
||||
return 1.0 / (1.0 + exp_neg)
|
||||
|
||||
@staticmethod
|
||||
def update_theta(
|
||||
theta: float, sigma_sq: float, outcome: float, b: float
|
||||
) -> tuple[float, float]:
|
||||
"""Bayesian update of theta given a binary (0/1) outcome.
|
||||
|
||||
Uses the standard 1PL Gaussian-approximation (Kalman-like) update:
|
||||
P = P_success(theta, b)
|
||||
new_precision = 1/sigma_sq + P*(1-P)
|
||||
new_sigma_sq = 1 / new_precision
|
||||
new_theta = theta + new_sigma_sq * (outcome - P)
|
||||
"""
|
||||
p = IRTEngine.P_success(theta, b)
|
||||
prior_precision = 1.0 / sigma_sq
|
||||
info = p * (1.0 - p)
|
||||
new_precision = prior_precision + info
|
||||
new_sigma_sq = 1.0 / new_precision
|
||||
new_theta = theta + new_sigma_sq * (outcome - p)
|
||||
return new_theta, new_sigma_sq
|
||||
|
||||
@staticmethod
|
||||
def select_scenario(
|
||||
theta: float,
|
||||
library: ScenarioLibrary,
|
||||
path: str,
|
||||
target_p: float = 0.7,
|
||||
observations: int = 0,
|
||||
) -> Scenario | None:
|
||||
"""Select the next scenario for a learner.
|
||||
|
||||
If observations < COLD_START_MIN_OBSERVATIONS (R-IRT-01), fall back to
|
||||
difficulty-based selection: pick the scenario whose `difficulty` is
|
||||
closest to round(theta + logit(target_p)).
|
||||
|
||||
Otherwise delegate to library.select_for_theta (IRT-aware selection
|
||||
targeting ~target_p).
|
||||
"""
|
||||
if observations < COLD_START_MIN_OBSERVATIONS:
|
||||
entries = library.list_by_path(path)
|
||||
if not entries:
|
||||
return None
|
||||
target_difficulty = round(theta + _logit(target_p))
|
||||
target_difficulty = max(1, min(5, target_difficulty))
|
||||
best_entry = None
|
||||
best_dist = math.inf
|
||||
for e in entries:
|
||||
dist = abs(e.difficulty - target_difficulty)
|
||||
if dist < best_dist:
|
||||
best_dist = dist
|
||||
best_entry = e
|
||||
if best_entry is None:
|
||||
return None
|
||||
return library.get(best_entry.id)
|
||||
return library.select_for_theta(theta, path, target_p=target_p)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"IRTEngine",
|
||||
"COLD_START_MIN_OBSERVATIONS",
|
||||
"DEFAULT_THETA",
|
||||
"DEFAULT_SIGMA_SQ",
|
||||
]
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Mastery score + gate logic — deterministic (SLICE-03 TASK-03-03).
|
||||
|
||||
Weighted mean of per-criterion levels with a conjunctive floor (every criterion
|
||||
>= 2 AND scenario mean >= 3.0 to pass). Path score is the mean over passing
|
||||
scenarios only. Gate opens at >=3 distinct passed scenarios AND path score
|
||||
>= 3.5 (D-032).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from server.mastery.rubric_scorer import CriterionScore
|
||||
from server.mastery.rubric_schema import Rubric
|
||||
|
||||
_SCENARIO_PASS_MEAN = 3.0
|
||||
_CONJUNCTIVE_FLOOR = 2
|
||||
_GATE_REQUIRED_DISTINCT = 3
|
||||
_GATE_REQUIRED_SCORE = 3.5
|
||||
|
||||
|
||||
class ScenarioScore(BaseModel):
|
||||
criterion_scores: list[CriterionScore]
|
||||
weighted_mean: float
|
||||
passed: bool
|
||||
fail_reason: str | None = None
|
||||
|
||||
@property
|
||||
def scenario_id(self) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
def compute_scenario_score(
|
||||
criterion_scores: list[CriterionScore], rubric: Rubric
|
||||
) -> ScenarioScore:
|
||||
"""Compute a deterministic scenario score with conjunctive-floor enforcement.
|
||||
|
||||
Pass requires: weighted mean >= 3.0 AND every criterion >= 2 AND any
|
||||
criterion with `conjunctive_floor` set must be >= that floor.
|
||||
"""
|
||||
weights = {c.id: c.weight for c in rubric.criteria}
|
||||
total = 0.0
|
||||
for cs in criterion_scores:
|
||||
w = weights.get(cs.criterion_id, cs.weight)
|
||||
total += cs.level * w
|
||||
mean = round(total, 6)
|
||||
|
||||
floor_violations: list[str] = []
|
||||
for cs in criterion_scores:
|
||||
c = rubric.criterion_by_id(cs.criterion_id)
|
||||
floor = c.conjunctive_floor if c else None
|
||||
required = max(floor or _CONJUNCTIVE_FLOOR, _CONJUNCTIVE_FLOOR)
|
||||
if cs.level < required:
|
||||
floor_violations.append(cs.criterion_id)
|
||||
|
||||
fail_reason: str | None = None
|
||||
if floor_violations:
|
||||
fail_reason = f"conjunctive_floor_violation:{','.join(floor_violations)}"
|
||||
elif mean < _SCENARIO_PASS_MEAN:
|
||||
fail_reason = f"mean_below_threshold:{mean}<{_SCENARIO_PASS_MEAN}"
|
||||
|
||||
passed = fail_reason is None
|
||||
return ScenarioScore(
|
||||
criterion_scores=criterion_scores,
|
||||
weighted_mean=mean,
|
||||
passed=passed,
|
||||
fail_reason=fail_reason,
|
||||
)
|
||||
|
||||
|
||||
def compute_path_score(passing_scenario_scores: list[ScenarioScore]) -> float:
|
||||
"""Mean weighted-mean over passing scenarios only. Empty → 0.0."""
|
||||
if not passing_scenario_scores:
|
||||
return 0.0
|
||||
return round(sum(s.weighted_mean for s in passing_scenario_scores) / len(passing_scenario_scores), 6)
|
||||
|
||||
|
||||
def check_gate(
|
||||
path_score: float,
|
||||
distinct_passed_count: int,
|
||||
*,
|
||||
required: int = _GATE_REQUIRED_DISTINCT,
|
||||
threshold: float = _GATE_REQUIRED_SCORE,
|
||||
) -> bool:
|
||||
"""Gate opens at >= `required` distinct passed scenarios AND path_score >= `threshold` (D-032)."""
|
||||
return distinct_passed_count >= required and path_score >= threshold
|
||||
|
||||
|
||||
__all__ = [
|
||||
"ScenarioScore",
|
||||
"compute_scenario_score",
|
||||
"compute_path_score",
|
||||
"check_gate",
|
||||
]
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Rubric loader — YAML → Pydantic Rubric (SLICE-01, D-039).
|
||||
|
||||
Loads a competency rubric by skill name from the `rubrics/` directory, validates
|
||||
it against the Pydantic schema, and caches the parsed result in-memory for the
|
||||
lifetime of the process. Used by the scoring engine (SLICE-03) and the path
|
||||
engine (SLICE-05).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from threading import Lock
|
||||
from typing import Dict
|
||||
|
||||
import yaml
|
||||
|
||||
from server.mastery.rubric_schema import Rubric, ValidationError
|
||||
|
||||
_DEFAULT_RUBRICS_DIR = Path(__file__).resolve().parent.parent.parent / "rubrics"
|
||||
|
||||
_cache: Dict[str, Rubric] = {}
|
||||
_cache_lock = Lock()
|
||||
|
||||
|
||||
def load_rubric(skill: str, rubrics_dir: Path | None = None) -> Rubric:
|
||||
"""Load and validate a rubric by skill name.
|
||||
|
||||
Args:
|
||||
skill: e.g. 'customer_service' (the YAML filename stem under rubrics/).
|
||||
rubrics_dir: override the rubrics directory (default: repo /rubrics).
|
||||
|
||||
Returns:
|
||||
A validated Rubric object. Cached in-memory per skill.
|
||||
|
||||
Raises:
|
||||
FileNotFoundError: if the YAML file doesn't exist.
|
||||
ValidationError: if the YAML fails schema validation (typed Pydantic error).
|
||||
"""
|
||||
with _cache_lock:
|
||||
cached = _cache.get(skill)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
base = rubrics_dir or _DEFAULT_RUBRICS_DIR
|
||||
path = base / f"{skill}.yaml"
|
||||
if not path.exists():
|
||||
raise FileNotFoundError(f"Rubric YAML not found: {skill} in {base}")
|
||||
|
||||
with path.open("r", encoding="utf-8") as f:
|
||||
raw = yaml.safe_load(f)
|
||||
|
||||
rubric = Rubric.model_validate(raw)
|
||||
|
||||
with _cache_lock:
|
||||
_cache[skill] = rubric
|
||||
return rubric
|
||||
|
||||
|
||||
def clear_cache() -> None:
|
||||
"""Clear the in-memory rubric cache (test helper)."""
|
||||
with _cache_lock:
|
||||
_cache.clear()
|
||||
|
||||
|
||||
__all__ = ["load_rubric", "clear_cache", "ValidationError"]
|
||||
@@ -0,0 +1,115 @@
|
||||
"""Praxis competency rubric schema — YAML → Pydantic (SLICE-01, D-039).
|
||||
|
||||
Defines the typed model for a competency rubric: 4+ criteria, each with 5
|
||||
behavioral anchor levels (Dreyfus + Miller "Does" + EPA entrustment per
|
||||
RESEARCH §2). Loaded from `rubrics/<skill>.yaml` by rubric_loader.py and
|
||||
referenced by the scoring engine (SLICE-03).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field, ValidationError, field_validator, model_validator
|
||||
|
||||
_LEVEL_FLOOR = 1
|
||||
_LEVEL_CEIL = 5
|
||||
_REQUIRED_LEVELS = 5
|
||||
_WEIGHT_TOLERANCE = 1e-6
|
||||
|
||||
|
||||
class RubricLevel(BaseModel):
|
||||
"""One anchor level (1=fail … 5=mastery/entrustable)."""
|
||||
|
||||
level: int = Field(..., ge=_LEVEL_FLOOR, le=_LEVEL_CEIL, description="1-5 level")
|
||||
label: str = Field(..., description="Short human label, e.g. 'Fail', 'Mastery / Entrustable'")
|
||||
anchor: str = Field(..., description="Observable-behavior anchor text (transcript-grounded)")
|
||||
signals: list[str] = Field(
|
||||
..., min_length=1, description="Observable behavior tags that map evidence to this level"
|
||||
)
|
||||
|
||||
|
||||
class RubricCriterion(BaseModel):
|
||||
"""One scoring criterion (e.g. empathy) with weight + 5 anchor levels."""
|
||||
|
||||
id: str = Field(..., description="Criterion id, e.g. 'empathy'")
|
||||
name: str = Field(..., description="Human-readable criterion name")
|
||||
weight: float = Field(..., ge=0.0, le=1.0, description="Criterion weight (sums to 1.0 across criteria)")
|
||||
conjunctive_floor: int | None = Field(
|
||||
None,
|
||||
ge=_LEVEL_FLOOR,
|
||||
le=_LEVEL_CEIL,
|
||||
description="If set, scenario cannot pass unless this criterion ≥ floor (professionalism ≥2)",
|
||||
)
|
||||
levels: list[RubricLevel] = Field(..., min_length=_REQUIRED_LEVELS, max_length=_REQUIRED_LEVELS)
|
||||
|
||||
@field_validator("levels")
|
||||
@classmethod
|
||||
def _levels_are_sequential(cls, v: list[RubricLevel]) -> list[RubricLevel]:
|
||||
seen = sorted(lvl.level for lvl in v)
|
||||
expected = list(range(_LEVEL_FLOOR, _LEVEL_CEIL + 1))
|
||||
if seen != expected:
|
||||
raise ValueError(
|
||||
f"criterion levels must be exactly 1..{_REQUIRED_LEVELS}, got {seen}"
|
||||
)
|
||||
return v
|
||||
|
||||
def level_by_value(self, level: int) -> RubricLevel | None:
|
||||
for lvl in self.levels:
|
||||
if lvl.level == level:
|
||||
return lvl
|
||||
return None
|
||||
|
||||
|
||||
class Rubric(BaseModel):
|
||||
"""A competency rubric for a skill (e.g. customer_service)."""
|
||||
|
||||
id: str = Field(..., description="Rubric id, e.g. 'customer_service'")
|
||||
skill: str = Field(..., description="Skill path this rubric scores, e.g. 'customer_service'")
|
||||
description: str | None = Field(None, description="Optional human description")
|
||||
criteria: list[RubricCriterion] = Field(..., min_length=1)
|
||||
archetype_weights: dict[str, dict[str, float]] | None = Field(
|
||||
None, description="Per-archetype weight overrides (D-039 amendment)"
|
||||
)
|
||||
escalated_weights: dict[str, float] | None = Field(
|
||||
None, description="Optional re-weight set when the escalate branch triggers (RESEARCH §6.3)"
|
||||
)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _validate_weights_and_ids(self) -> Rubric:
|
||||
total = sum(c.weight for c in self.criteria)
|
||||
if abs(total - 1.0) > _WEIGHT_TOLERANCE:
|
||||
raise ValueError(
|
||||
f"criterion weights must sum to 1.0 (±{_WEIGHT_TOLERANCE}), got {total}"
|
||||
)
|
||||
ids = [c.id for c in self.criteria]
|
||||
if len(ids) != len(set(ids)):
|
||||
dupes = sorted({i for i in ids if ids.count(i) > 1})
|
||||
raise ValueError(f"duplicate criterion ids: {dupes}")
|
||||
if self.skill != self.id and not self.id.startswith(self.skill):
|
||||
pass
|
||||
return self
|
||||
|
||||
def criterion_by_id(self, criterion_id: str) -> RubricCriterion | None:
|
||||
for c in self.criteria:
|
||||
if c.id == criterion_id:
|
||||
return c
|
||||
return None
|
||||
|
||||
def weights_for_archetype(self, archetype: str | None) -> dict[str, float]:
|
||||
"""Return {criterion_id: weight} for an archetype, falling back to the base weights."""
|
||||
if archetype and self.archetype_weights and archetype in self.archetype_weights:
|
||||
override = self.archetype_weights[archetype]
|
||||
return {c.id: override.get(c.id, c.weight) for c in self.criteria}
|
||||
return {c.id: c.weight for c in self.criteria}
|
||||
|
||||
def criterion_ids(self) -> list[str]:
|
||||
return [c.id for c in self.criteria]
|
||||
|
||||
|
||||
__all__ = [
|
||||
"Rubric",
|
||||
"RubricCriterion",
|
||||
"RubricLevel",
|
||||
"ValidationError",
|
||||
]
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Rule-based rubric scorer — deterministic (SLICE-03 TASK-03-02, REQ-NFR-MAST-01).
|
||||
|
||||
No LLM. Maps evidence signals to rubric level anchors: for each criterion, pick
|
||||
the highest level whose `signals[]` are all present in the matched evidence,
|
||||
fallback to level 1 if no level matches. The output is reproducible given the
|
||||
same (evidence, rubric) pair.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from server.mastery.evidence_extractor import Evidence
|
||||
from server.mastery.rubric_schema import Rubric, RubricCriterion
|
||||
|
||||
|
||||
class CriterionScore(BaseModel):
|
||||
criterion_id: str
|
||||
level: int = Field(ge=1, le=5)
|
||||
weight: float
|
||||
evidence_quote: str = ""
|
||||
matched_signals: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
def _evidence_for(evidence: list[Evidence], criterion_id: str) -> Evidence | None:
|
||||
for ev in evidence:
|
||||
if ev.criterion_id == criterion_id:
|
||||
return ev
|
||||
return None
|
||||
|
||||
|
||||
def _level_for_criterion(criterion: RubricCriterion, ev: Evidence | None) -> tuple[int, list[str]]:
|
||||
if ev is None or not ev.signals:
|
||||
return 1, []
|
||||
ev_signals = set(ev.signals)
|
||||
best_level = 1
|
||||
best_signals: list[str] = []
|
||||
for lvl in sorted(criterion.levels, key=lambda l: l.level):
|
||||
if all(s in ev_signals for s in lvl.signals):
|
||||
best_level = lvl.level
|
||||
best_signals = list(lvl.signals)
|
||||
return best_level, best_signals
|
||||
|
||||
|
||||
def score(evidence: list[Evidence], rubric: Rubric) -> list[CriterionScore]:
|
||||
"""Score evidence against the rubric — deterministic, no LLM.
|
||||
|
||||
Returns one CriterionScore per rubric criterion, in rubric order. Criteria
|
||||
with no matching evidence get level 1 (the "Fail" anchor).
|
||||
"""
|
||||
out: list[CriterionScore] = []
|
||||
for c in rubric.criteria:
|
||||
ev = _evidence_for(evidence, c.id)
|
||||
level, matched = _level_for_criterion(c, ev)
|
||||
out.append(
|
||||
CriterionScore(
|
||||
criterion_id=c.id,
|
||||
level=level,
|
||||
weight=c.weight,
|
||||
evidence_quote=ev.quote if ev else "",
|
||||
matched_signals=matched,
|
||||
)
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
__all__ = ["CriterionScore", "score"]
|
||||
Reference in New Issue
Block a user