Files
praxis/server/cohort/aggregator.py
T
Praxis CI ec397f2c65 docs(milestone): complete v0.5-live-assist — v0.1.13 tagged, milestone release, merged to main
v0.5 (Live Assist — on-the-job voice companion) milestone complete.
4 phases: P0 (pre-execution, v0.1.10) → P1 (assist core + guardrail,
v0.1.11) → P2 (integration + tech-debt + NFR, v0.1.12) → P3 (final
review + ship, v0.1.13 = milestone release).

16/16 REQs covered (3 ASSIST + 4 NFR + 9 IDEATE). 4 v0.6 backlog.
469 tests passed, 0 failed. 1 P0 fixed (guardrail processor safety).
8 P1+ flagged for v0.6. 8 v0.4 P1+ tech-debt addressed.
G-049 + G-067 grill MUSTs resolved. ESCALATION-01 (PIPEDA) OPEN for
human legal review before assist surface go-live.

---ci---
project: praxis
phase: 3
milestone: v0.5
status: complete
requirements:
  covered: [REQ-ASSIST-01, REQ-ASSIST-02, REQ-ASSIST-03, REQ-NFR-ASSIST-01, REQ-NFR-ASSIST-02, REQ-NFR-ASSIST-03, REQ-NFR-ASSIST-04, REQ-IDEATE-01, REQ-IDEATE-02, REQ-IDEATE-03, REQ-IDEATE-04, REQ-IDEATE-05, REQ-IDEATE-06, REQ-IDEATE-07, REQ-IDEATE-08, REQ-IDEATE-09]
  partial: []
---/ci---
2026-08-04 22:35:56 +00:00

406 lines
19 KiB
Python

"""Cohort aggregation logic + k-anonymity suppression (TASK-07-01, D-034, D-045).
Computes k-anonymized aggregates for the affected (path, metric, window_start)
bins and upserts them to cohort_aggregates via PgStore. Suppression is at
write time (auditable — RESEARCH-v0.4 §3.1): COUNT(DISTINCT learner_ref) < 10
=> cell_suppressed=TRUE, value=NULL.
Metrics computed (per 7-day rolling window, per path):
sessions_count, active_learners_count, gate_open_rate,
median_mastery_score, failure_mode_frequency,
rubric_criterion_means, week_distribution.
The session_outcome dict contains: learner_ref (opaque — D-031), path,
scenario_id, outcome (pass/fail), rubric_scores, failure_mode, branch_path,
timestamp.
No raw learner PII in Postgres (D-031): only aggregates + opaque learner_ref
for distinct counting.
"""
from __future__ import annotations
import datetime as _dt
import logging
import statistics
from typing import Any
from db.pg_store import PgStore
log = logging.getLogger(__name__)
K_ANON_THRESHOLD = 10
def _rolling_window(now: _dt.datetime | None = None) -> tuple[_dt.date, _dt.date]:
"""Return the 7-day rolling window (start, end) for `now`.
window_start = today - 6 days, window_end = today (inclusive 7-day span).
"""
today = (now or _dt.datetime.now(_dt.timezone.utc)).date()
return today - _dt.timedelta(days=6), today
def _distinct_learners(sessions: list[dict[str, Any]]) -> int:
return len({s["learner_ref"] for s in sessions if s.get("learner_ref")})
async def aggregate_session(pg_store: PgStore, session_outcome: dict[str, Any]) -> None:
"""Compute + upsert k-anonymized aggregates for one session outcome.
Branches on `session_type` (D-062):
- 'assist' → _aggregate_assist (assist metrics, no mastery — D-063)
- else → _aggregate_practice (the existing v0.4 practice logic)
Reads the affected path's recent session set (from cohort_aggregates or
an in-memory accumulator), recomputes the metric cells for the 7-day
window, applies k-anon suppression, and upserts each cell idempotently.
Idempotent (ON CONFLICT upsert) — re-running with the same outcome
produces the same aggregate. The caller (hook.py) passes one session at
a time; the nightly job (nightly.py) recomputes the full window.
"""
session_type = session_outcome.get("session_type", "practice")
if session_type == "assist":
await _aggregate_assist(pg_store, session_outcome)
else:
await _aggregate_practice(pg_store, session_outcome)
async def _aggregate_practice(pg_store: PgStore, session_outcome: dict[str, Any]) -> None:
"""The v0.4 practice aggregation logic (renamed for clarity — D-062).
Computes: sessions_count, active_learners_count, gate_open_rate,
median_mastery_score, failure_mode_frequency, rubric_criterion_means,
week_distribution. k-anon suppression (≥10 distinct learners).
"""
path = session_outcome.get("path") or session_outcome.get("path_id") or "unknown"
learner_ref = session_outcome.get("learner_ref") or "unknown"
outcome = session_outcome.get("outcome", "fail")
rubric_scores = session_outcome.get("rubric_scores") or []
failure_mode = session_outcome.get("failure_mode")
branch_path = session_outcome.get("branch_path") or []
scenario_id = session_outcome.get("scenario_id")
ts = session_outcome.get("timestamp")
window_start, window_end = _rolling_window(
_dt.datetime.fromisoformat(ts) if isinstance(ts, str) else None
)
# Distinct-learner count for k-anon: this session's learner + any others
# already recorded for the same (path, window). For the per-session hook
# we accumulate by appending to a sessions_count cell + tracking distinct
# learner_refs via active_learners_count. The nightly job recomputes from
# the mastery_gate_events + session log (full reconciliation).
#
# For the on-session-end hook we cannot cheaply know all distinct learners
# without a raw-events table (which we deliberately do not maintain for PII
# reasons — D-031). We instead maintain a single active_learners_count
# counter per (path, window) and the nightly job reconciles the true
# distinct count from mastery_gate_events. The hook uses the running
# counter; if it is < K_ANON_THRESHOLD we suppress.
active_count = await _bump_active_learners(pg_store, path, window_start, learner_ref)
sessions_count = await _bump_counter(pg_store, path, "sessions_count", window_start, window_end)
suppressed = active_count < K_ANON_THRESHOLD
await _upsert_cell(pg_store, path, "sessions_count", window_start, window_end,
float(sessions_count) if not suppressed else None,
active_count, suppressed)
await _upsert_cell(pg_store, path, "active_learners_count", window_start, window_end,
float(active_count) if not suppressed else None,
active_count, suppressed)
# gate_open_rate: 1.0 if this session passed, 0.0 otherwise (running mean
# reconciled by nightly). Stored as the fraction of pass outcomes seen.
passed = 1.0 if outcome == "pass" else 0.0
gate_open_rate = await _running_mean(pg_store, path, "gate_open_rate",
window_start, window_end, passed, active_count)
await _upsert_cell(pg_store, path, "gate_open_rate", window_start, window_end,
gate_open_rate if not suppressed else None,
active_count, suppressed)
# median_mastery_score (from rubric scores) — running median reconciled nightly
if rubric_scores:
scores = [float(r.get("score", r.get("weighted_mean", 0.0))) for r in rubric_scores]
scenario_mean = statistics.mean(scores) if scores else 0.0
median_val = await _running_mean(pg_store, path, "median_mastery_score",
window_start, window_end, scenario_mean, active_count)
await _upsert_cell(pg_store, path, "median_mastery_score", window_start, window_end,
median_val if not suppressed else None,
active_count, suppressed)
# rubric_criterion_means — one cell per criterion id
for r in rubric_scores:
cid = r.get("criterion_id") or r.get("id") or "unknown"
score = float(r.get("score", 0.0))
mean_val = await _running_mean(pg_store, path, f"rubric_criterion_mean:{cid}",
window_start, window_end, score, active_count)
await _upsert_cell(pg_store, path, f"rubric_criterion_mean:{cid}",
window_start, window_end,
mean_val if not suppressed else None,
active_count, suppressed)
# failure_mode_frequency — one cell per observed mode
if failure_mode:
freq = await _bump_mode_counter(pg_store, path, f"failure_mode:{failure_mode}",
window_start, window_end)
await _upsert_cell(pg_store, path, f"failure_mode:{failure_mode}",
window_start, window_end,
float(freq) if not suppressed else None,
active_count, suppressed)
# week_distribution — branch_path captures the path-week; record one cell
# per branch outcome seen.
if branch_path:
last_branch = branch_path[-1] if isinstance(branch_path, list) else str(branch_path)
freq = await _bump_mode_counter(pg_store, path, f"branch:{last_branch}",
window_start, window_end)
await _upsert_cell(pg_store, path, f"branch:{last_branch}",
window_start, window_end,
float(freq) if not suppressed else None,
active_count, suppressed)
log.debug(
"aggregate_session path=%s learner=%s outcome=%s window=%s..%s "
"active=%d suppressed=%s",
path, learner_ref, outcome, window_start, window_end,
active_count, suppressed,
)
# ── Assist aggregation (D-062, D-063, TASK-10-01) ────────────────────────────
# Assist metrics use the SAME k-anonymity suppression (≥10 distinct learners),
# the SAME 7-day rolling window, + the SAME idempotent upsert as practice.
# No schema change to cohort_aggregates (the `metric` column is free-form TEXT
# — D-062). D-063: assist does NOT update mastery (no rubric scores, no
# gate_open_rate — those are practice-only metrics).
# The 5 core assist metrics (REQ-NFR-ASSIST-04) + p95 latency + cost:
# assist_shifts_count — count of assist shifts in the window
# assist_turns_count — total assist turns across all shifts
# assist_avg_turns_per_shift — running mean of turns per shift
# assist_active_learners_count — distinct learners with assist shifts
# assist_guardrail_block_rate — guardrail_blocks / assist_turns_count
# assist_p95_latency_ms — D-072 p95 latency (from SLICE-09)
# assist_avg_cost_per_shift — per-shift assist cost (from SLICE-11, optional)
async def _aggregate_assist(pg_store: PgStore, session_outcome: dict[str, Any]) -> None:
"""Aggregate one assist shift outcome (D-062, D-063, TASK-10-01).
Upserts the 5 core assist metrics + p95 latency (+ optional avg cost).
k-anon suppression applies (≥10 distinct learners — D-034 carry-forward).
Idempotent upsert (ON CONFLICT). No schema change (D-062 — metric is TEXT).
D-063: assist does NOT update mastery. This function computes NO mastery
metrics (no rubric scores, no gate_open_rate). The practice branch owns
mastery; the assist branch owns assist-only metrics.
"""
path = session_outcome.get("path") or session_outcome.get("path_id") or "unknown"
learner_ref = session_outcome.get("learner_ref") or "unknown"
turn_count = int(session_outcome.get("assist_turn_count", 0))
blocks = int(session_outcome.get("guardrail_blocks", 0))
p95_latency = session_outcome.get("assist_p95_latency_ms")
p95_latency_f = float(p95_latency) if p95_latency is not None else None
cost_cents = int(session_outcome.get("assist_cost_cents", 0) or 0)
ts = session_outcome.get("timestamp")
window_start, window_end = _rolling_window(
_dt.datetime.fromisoformat(ts) if isinstance(ts, str) else None
)
# Distinct-learner count for k-anon (same in-memory cache as practice).
active_count = await _bump_active_learners(pg_store, path, window_start, learner_ref)
shifts_count = await _bump_counter(pg_store, path, "assist_shifts_count",
window_start, window_end)
turns_total = await _bump_assist_turns(pg_store, path, window_start, turn_count)
suppressed = active_count < K_ANON_THRESHOLD
# assist_shifts_count
await _upsert_cell(pg_store, path, "assist_shifts_count", window_start, window_end,
float(shifts_count) if not suppressed else None,
active_count, suppressed)
# assist_active_learners_count
await _upsert_cell(pg_store, path, "assist_active_learners_count",
window_start, window_end,
float(active_count) if not suppressed else None,
active_count, suppressed)
# assist_turns_count
await _upsert_cell(pg_store, path, "assist_turns_count", window_start, window_end,
float(turns_total) if not suppressed else None,
active_count, suppressed)
# assist_avg_turns_per_shift — running mean of turns per shift
avg_turns = await _running_mean(pg_store, path, "assist_avg_turns_per_shift",
window_start, window_end, float(turn_count),
active_count)
await _upsert_cell(pg_store, path, "assist_avg_turns_per_shift",
window_start, window_end,
avg_turns if not suppressed else None,
active_count, suppressed)
# assist_guardrail_block_rate = blocks / turns (0 if no turns yet)
block_rate = (blocks / turn_count) if turn_count > 0 else 0.0
# Running mean of per-shift block rates (so the window value is the mean
# across shifts, not just the latest shift's rate).
avg_block_rate = await _running_mean(pg_store, path, "assist_guardrail_block_rate",
window_start, window_end, block_rate,
active_count)
await _upsert_cell(pg_store, path, "assist_guardrail_block_rate",
window_start, window_end,
avg_block_rate if not suppressed else None,
active_count, suppressed)
# assist_p95_latency_ms (D-072 — from SLICE-09). Running mean of per-shift
# p95 so the window value is the mean p95 across shifts (a trend signal).
if p95_latency_f is not None:
avg_p95 = await _running_mean(pg_store, path, "assist_p95_latency_ms",
window_start, window_end, p95_latency_f,
active_count)
await _upsert_cell(pg_store, path, "assist_p95_latency_ms",
window_start, window_end,
avg_p95 if not suppressed else None,
active_count, suppressed)
# assist_avg_cost_per_shift (TASK-11-01 — optional, useful for C-3 check).
# Running mean of per-shift cost in cents.
if cost_cents > 0:
avg_cost = await _running_mean(pg_store, path, "assist_avg_cost_per_shift",
window_start, window_end, float(cost_cents),
active_count)
await _upsert_cell(pg_store, path, "assist_avg_cost_per_shift",
window_start, window_end,
avg_cost if not suppressed else None,
active_count, suppressed)
log.debug(
"aggregate_assist path=%s learner=%s turns=%d blocks=%d p95=%s "
"window=%s..%s active=%d suppressed=%s",
path, learner_ref, turn_count, blocks, p95_latency_f,
window_start, window_end, active_count, suppressed,
)
# ── Internal cell upsert + counter helpers ──────────────────────────────────
# The PgStore.upsert_cohort_aggregate is idempotent (ON CONFLICT). We use a
# small in-memory cache on the PgStore instance (created lazily) to track
# per-(path, metric, window) running counters + distinct learner sets. The
# nightly job bypasses this cache and recomputes from mastery_gate_events.
def _cache(pg_store: PgStore) -> dict:
cache = getattr(pg_store, "_agg_cache", None)
if not isinstance(cache, dict):
cache = {}
try:
pg_store._agg_cache = cache # type: ignore[attr-defined]
except Exception:
pass
return cache
def _ck(path: str, metric: str, window_start: _dt.date) -> tuple:
return (path, metric, window_start)
async def _upsert_cell(pg_store: PgStore, path: str, metric: str,
window_start: _dt.date, window_end: _dt.date,
value: float | None, cell_count: int,
suppressed: bool) -> None:
await pg_store.upsert_cohort_aggregate(
path, metric, window_start, window_end, value, cell_count, suppressed,
)
async def _bump_active_learners(pg_store: PgStore, path: str,
window_start: _dt.date, learner_ref: str) -> int:
"""Track distinct learner_refs per (path, window) in the in-memory cache.
Returns the current distinct count (after adding this learner). The
nightly job reconciles the true count from mastery_gate_events.
TASK-12-01 (P1+ #7): on the first call for a (path, window), the in-memory
cache is seeded from the persisted SQLite cache (cohort_learner_cache) so
the distinct count survives a server restart. The cache is persisted
periodically via _save_learner_cache() (called by the hook on shift-end).
"""
cache = _cache(pg_store)
key = _ck(path, "__learners__", window_start)
learners: set[str] = cache.get(key)
if learners is None:
# First call for this (path, window) since restart → seed from the
# persisted SQLite cache (TASK-12-01). If the cache is empty (fresh
# install or first run), this starts a new set.
try:
from server.cohort.learner_cache import _count_distinct_learners, _load_learner_cache
persisted = await _load_learner_cache(pg_store)
# Merge any persisted learners for this (path, window).
learners = persisted.get(key, set()).copy()
except Exception:
log.debug("cohort_learner_cache: load failed (fresh start?) — using empty set")
learners = set()
learners.add(learner_ref)
cache[key] = learners
# Persist the updated set to SQLite (TASK-12-01 — survives restart).
try:
from server.cohort.learner_cache import _save_learner_cache
await _save_learner_cache(pg_store, {key: learners})
except Exception:
log.debug("cohort_learner_cache: save failed (non-fatal — nightly reconciles)")
return len(learners)
async def _bump_counter(pg_store: PgStore, path: str, metric: str,
window_start: _dt.date, window_end: _dt.date) -> int:
cache = _cache(pg_store)
key = _ck(path, metric, window_start)
cache[key] = cache.get(key, 0) + 1
return cache[key]
async def _bump_mode_counter(pg_store: PgStore, path: str, metric: str,
window_start: _dt.date, window_end: _dt.date) -> int:
return await _bump_counter(pg_store, path, metric, window_start, window_end)
async def _bump_assist_turns(pg_store: PgStore, path: str,
window_start: _dt.date, turn_count: int) -> int:
"""Accumulate assist turns across shifts in the window (TASK-10-01).
The counter is a running total of assist turns across all shifts in the
(path, window). Each shift contributes its `assist_turn_count`.
"""
cache = _cache(pg_store)
key = _ck(path, "assist_turns_count", window_start)
cache[key] = cache.get(key, 0) + int(turn_count)
return cache[key]
async def _running_mean(pg_store: PgStore, path: str, metric: str,
window_start: _dt.date, window_end: _dt.date,
value: float, _active_count: int) -> float:
"""Incremental running mean per (path, metric, window)."""
cache = _cache(pg_store)
k = _ck(path, metric, window_start)
n_key = _ck(path, metric + "__n__", window_start)
n = cache.get(n_key, 0)
prev = cache.get(k, 0.0)
new_n = n + 1
new_mean = prev + (value - prev) / new_n
cache[k] = new_mean
cache[n_key] = new_n
return new_mean
__all__ = [
"aggregate_session",
"_aggregate_practice",
"_aggregate_assist",
"K_ANON_THRESHOLD",
"_rolling_window",
]