ec397f2c65
v0.5 (Live Assist — on-the-job voice companion) milestone complete. 4 phases: P0 (pre-execution, v0.1.10) → P1 (assist core + guardrail, v0.1.11) → P2 (integration + tech-debt + NFR, v0.1.12) → P3 (final review + ship, v0.1.13 = milestone release). 16/16 REQs covered (3 ASSIST + 4 NFR + 9 IDEATE). 4 v0.6 backlog. 469 tests passed, 0 failed. 1 P0 fixed (guardrail processor safety). 8 P1+ flagged for v0.6. 8 v0.4 P1+ tech-debt addressed. G-049 + G-067 grill MUSTs resolved. ESCALATION-01 (PIPEDA) OPEN for human legal review before assist surface go-live. ---ci--- project: praxis phase: 3 milestone: v0.5 status: complete requirements: covered: [REQ-ASSIST-01, REQ-ASSIST-02, REQ-ASSIST-03, REQ-NFR-ASSIST-01, REQ-NFR-ASSIST-02, REQ-NFR-ASSIST-03, REQ-NFR-ASSIST-04, REQ-IDEATE-01, REQ-IDEATE-02, REQ-IDEATE-03, REQ-IDEATE-04, REQ-IDEATE-05, REQ-IDEATE-06, REQ-IDEATE-07, REQ-IDEATE-08, REQ-IDEATE-09] partial: [] ---/ci---
208 lines
7.4 KiB
Python
208 lines
7.4 KiB
Python
"""AssistContextBinder — loads path week + scenario tag + learner theta from
|
|
SQLite into a ≤150-token assist system prompt (D-059, D-066, TASK-01-02).
|
|
|
|
The context string is terse by design (D-066): the coaching instruction is a
|
|
fixed ~80-token block; the context-binding is a per-shift ~50-token block; the
|
|
voice-conciseness tail is ~20 tokens. Total ≤200 words (rough word≈token check
|
|
— the real token count is verified in the pipeline test).
|
|
|
|
Missing learner state (no progress row, no theta) → defaults are used
|
|
(week=1, theta=0.0, focus=generic). The prompt is never empty.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import yaml
|
|
|
|
from db.store import PraxisStore, HARDCODED_LEARNER_ID
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
_DEFAULT_PATHS_DIR = Path(__file__).resolve().parent.parent.parent / "paths"
|
|
|
|
# Layer 1 — coaching instruction (~80 tokens, fixed). Same text as the
|
|
# LiveAssistGuardrail.session_start_disclaimer (D-066). The disclaimer is NOT
|
|
# played as audio at shift start (unlike practice) — it's the system-prompt
|
|
# prefix. The consent disclosure (server/assist/consent.py) is separate.
|
|
COACHING_INSTRUCTION = (
|
|
"You are a live coaching AI in the learner's ear during a real customer "
|
|
"interaction. Coach, do not do the learner's job. Ask guiding questions; "
|
|
"never give the answer. Never speak on behalf of the learner. Never claim "
|
|
"authority you don't have. Keep responses to 1-3 sentences for voice."
|
|
)
|
|
|
|
# Voice-conciseness tail (~20 tokens, fixed).
|
|
VOICE_CONCISENESS = "Be brief. The customer is waiting."
|
|
|
|
# Default coaching focus when no rubric data is available.
|
|
_DEFAULT_COACHING_FOCUS = "empathy + resolution-concreteness"
|
|
|
|
# Rough word budget (D-066 — ≤150 tokens; word≈token is a conservative upper
|
|
# bound since English averages ~1.3 tokens/word). 200 words ≈ 150-260 tokens.
|
|
_MAX_PROMPT_WORDS = 200
|
|
|
|
|
|
@dataclass
|
|
class AssistContext:
|
|
"""The bound context for one assist shift (TASK-01-02)."""
|
|
|
|
system_prompt: str
|
|
current_week: int
|
|
scenario_tag: str
|
|
theta: float
|
|
coaching_focus: str
|
|
path_slug: str
|
|
|
|
|
|
def _week_focus(path_slug: str, week: int) -> str:
|
|
"""Derive the week focus string from the path YAML (D-059)."""
|
|
path_file = _DEFAULT_PATHS_DIR / f"{path_slug}.yaml"
|
|
if not path_file.exists():
|
|
return f"Week {week}"
|
|
try:
|
|
with path_file.open("r", encoding="utf-8") as f:
|
|
path_doc = yaml.safe_load(f) or {}
|
|
weeks = path_doc.get("weeks") or []
|
|
# weeks is 1-indexed in the YAML; list is 0-indexed.
|
|
if 1 <= week <= len(weeks):
|
|
entry = weeks[week - 1]
|
|
title = entry.get("title") if isinstance(entry, dict) else None
|
|
if title:
|
|
return title
|
|
return f"Week {week}"
|
|
except Exception:
|
|
log.warning("failed to read path YAML %s; defaulting week focus", path_file)
|
|
return f"Week {week}"
|
|
|
|
|
|
def _top_rubric_criterion(
|
|
store: PraxisStore, learner_id: str, path_slug: str
|
|
) -> str:
|
|
"""Sync fallback for the coaching focus (unused — kept for reference).
|
|
|
|
The async path (_async_top_rubric_criterion) is what bind() actually calls.
|
|
"""
|
|
return _DEFAULT_COACHING_FOCUS
|
|
|
|
|
|
class AssistContextBinder:
|
|
"""Loads context for an assist shift from SQLite + scenario library.
|
|
|
|
D-059: learner declares context (path week + scenario tag) at shift start;
|
|
the server reads progress.current_week + theta from SQLite for rubric
|
|
alignment + coaching focus.
|
|
"""
|
|
|
|
def __init__(self, store: PraxisStore) -> None:
|
|
self.store = store
|
|
|
|
async def bind(
|
|
self,
|
|
learner_id: str,
|
|
path_slug: str,
|
|
scenario_tag: str,
|
|
) -> AssistContext:
|
|
"""Construct the ≤150-token assist system prompt for this shift."""
|
|
# Read learner state from SQLite (D-007). Missing → defaults.
|
|
current_week = 1
|
|
theta = 0.0
|
|
try:
|
|
progress = await self.store.get_progress(learner_id, path_slug)
|
|
if progress is not None:
|
|
current_week = int(progress.get("current_week", 1) or 1)
|
|
except Exception:
|
|
log.warning("get_progress failed for %s/%s; defaulting week=1", learner_id, path_slug)
|
|
|
|
try:
|
|
ability = await self.store.get_ability(learner_id, path_slug)
|
|
if ability is not None:
|
|
theta = float(ability.get("theta", 0.0) or 0.0)
|
|
except Exception:
|
|
log.warning("get_ability failed for %s/%s; defaulting theta=0.0", learner_id, path_slug)
|
|
|
|
# Coaching focus = the learner's weakest rubric criterion.
|
|
coaching_focus = await self._async_top_rubric_criterion(learner_id, path_slug)
|
|
week_focus = _week_focus(path_slug, current_week)
|
|
|
|
# Context-binding block (~50 tokens, per shift).
|
|
context_binding = (
|
|
f"Week {current_week}: {week_focus}. Scenario: {scenario_tag}. "
|
|
f"Learner theta: {theta:.1f}. Coaching focus: {coaching_focus}."
|
|
)
|
|
|
|
system_prompt = (
|
|
f"{COACHING_INSTRUCTION}\n\n"
|
|
f"{context_binding}\n\n"
|
|
f"{VOICE_CONCISENESS}"
|
|
)
|
|
|
|
# Token-budget assertion (rough word≈token check; D-066).
|
|
word_count = len(system_prompt.split())
|
|
if word_count > _MAX_PROMPT_WORDS:
|
|
log.warning(
|
|
"assist system prompt exceeds %d words (%d) — truncating context-binding (D-066)",
|
|
_MAX_PROMPT_WORDS, word_count,
|
|
)
|
|
# Truncate the context-binding section to fit the budget.
|
|
system_prompt = (
|
|
f"{COACHING_INSTRUCTION}\n\n"
|
|
f"Week {current_week}, {scenario_tag}.\n\n"
|
|
f"{VOICE_CONCISENESS}"
|
|
)
|
|
|
|
return AssistContext(
|
|
system_prompt=system_prompt,
|
|
current_week=current_week,
|
|
scenario_tag=scenario_tag,
|
|
theta=theta,
|
|
coaching_focus=coaching_focus,
|
|
path_slug=path_slug,
|
|
)
|
|
|
|
async def _async_top_rubric_criterion(
|
|
self, learner_id: str, path_slug: str
|
|
) -> str:
|
|
"""Async version of _top_rubric_criterion (calls store directly)."""
|
|
try:
|
|
events = await self.store.list_gate_events(learner_id, path_slug)
|
|
except Exception:
|
|
events = []
|
|
if not events:
|
|
return _DEFAULT_COACHING_FOCUS
|
|
import json
|
|
|
|
sums: dict[str, float] = {}
|
|
counts: dict[str, int] = {}
|
|
for ev in events:
|
|
scores_json = ev.get("rubric_scores_json")
|
|
if isinstance(scores_json, str):
|
|
try:
|
|
scores = json.loads(scores_json)
|
|
except Exception:
|
|
continue
|
|
elif isinstance(scores_json, list):
|
|
scores = scores_json
|
|
else:
|
|
continue
|
|
for s in scores:
|
|
cid = s.get("criterion_id") or s.get("id") or "unknown"
|
|
score = float(s.get("score", 0.0))
|
|
sums[cid] = sums.get(cid, 0.0) + score
|
|
counts[cid] = counts.get(cid, 0) + 1
|
|
if not counts:
|
|
return _DEFAULT_COACHING_FOCUS
|
|
means = {cid: sums[cid] / counts[cid] for cid in counts}
|
|
return min(means, key=means.get) # type: ignore[arg-type]
|
|
|
|
|
|
__all__ = [
|
|
"AssistContextBinder",
|
|
"AssistContext",
|
|
"COACHING_INSTRUCTION",
|
|
"VOICE_CONCISENESS",
|
|
] |