7b1b296430
client/ — React + Vite + TypeScript scaffolded with the Pipecat client SDK (@pipecat-ai/client-js) and SmallWebRTCTransport (@pipecat-ai/small-webrtc-transport). useVoiceSession.ts hook manages mic permission, WebRTC connect, audio playback, live transcript, and a latency readout (captures the e2e_latency_ms metric the server emits). App.tsx is a minimal one-page session UI: disclaimer, Start/End buttons, status badge, latency readout (within/over 600ms budget), live transcript. vite.config.ts proxies /pipecat + /health to the Python server (port 8789). npm run typecheck + npm run build pass. server/latency.py — LatencyObserver (a Pipecat FrameProcessor) timestamps transcript-ready, LLM-first-token, TTS-first-audio, and playback-start per turn, computes ASR→TTS-first-audio (the v0.1 latency target), and logs it to console with a within/over-budget verdict. Wired into the pipeline between STT/LLM/TTS so it observes without altering the frame stream. 5 unit tests pass (LatencyRecord e2e math + observer construction + reset_turn). Full server suite: 18 passed. ---ci--- phase: 1 milestone: v0.1 plan: 02 task: 02-05,02-06 status: execute persona: frontend-engineer,backend-engineer requirements: covered: [REQ-VOICE-01, REQ-VOICE-02, REQ-VOICE-03, REQ-NFR-LAT-01] ---/ci---
48 lines
1.5 KiB
Python
48 lines
1.5 KiB
Python
"""Unit tests for the LatencyObserver (TASK-02-06)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from server.latency import LatencyObserver, LatencyRecord
|
|
|
|
|
|
def test_latency_record_e2e():
|
|
r = LatencyRecord(transcript_ready_ms=100.0, tts_first_audio_ms=650.0)
|
|
assert r.e2e_asr_to_tts_ms == 550.0
|
|
|
|
|
|
def test_latency_record_missing_segments():
|
|
r = LatencyRecord(transcript_ready_ms=100.0)
|
|
assert r.e2e_asr_to_tts_ms is None
|
|
r2 = LatencyRecord(tts_first_audio_ms=650.0)
|
|
assert r2.e2e_asr_to_tts_ms is None
|
|
|
|
|
|
def test_latency_record_metric_dict():
|
|
r = LatencyRecord(
|
|
transcript_ready_ms=100.0, llm_first_token_ms=300.0, tts_first_audio_ms=650.0
|
|
)
|
|
m = r.as_metric()
|
|
assert m["e2e_latency_ms"] == 550.0
|
|
assert m["llm_first_token_ms"] == 300.0
|
|
|
|
|
|
def test_latency_observer_construction():
|
|
"""The observer constructs cleanly and starts with an empty state."""
|
|
obs = LatencyObserver()
|
|
assert obs.state.records == []
|
|
assert obs.state.current.transcript_ready_ms is None
|
|
|
|
|
|
def test_latency_observer_state_reset_turn():
|
|
"""reset_turn archives the current record and starts a fresh one."""
|
|
from server.latency import LatencyObserverState
|
|
|
|
state = LatencyObserverState()
|
|
state.current.transcript_ready_ms = 100.0
|
|
state.current.tts_first_audio_ms = 650.0
|
|
state.reset_turn()
|
|
assert len(state.records) == 1
|
|
assert state.records[0].e2e_asr_to_tts_ms == 550.0
|
|
assert state.current.transcript_ready_ms is None |