Files
praxis/server/pipeline.py
T
Praxis CI 7b1b296430 feat(P01-02-05,P01-02-06): React client + latency readout
client/ — React + Vite + TypeScript scaffolded with the Pipecat client SDK
(@pipecat-ai/client-js) and SmallWebRTCTransport
(@pipecat-ai/small-webrtc-transport). useVoiceSession.ts hook manages mic
permission, WebRTC connect, audio playback, live transcript, and a latency
readout (captures the e2e_latency_ms metric the server emits). App.tsx is a
minimal one-page session UI: disclaimer, Start/End buttons, status badge,
latency readout (within/over 600ms budget), live transcript. vite.config.ts
proxies /pipecat + /health to the Python server (port 8789). npm run
typecheck + npm run build pass.

server/latency.py — LatencyObserver (a Pipecat FrameProcessor) timestamps
transcript-ready, LLM-first-token, TTS-first-audio, and playback-start per
turn, computes ASR→TTS-first-audio (the v0.1 latency target), and logs it
to console with a within/over-budget verdict. Wired into the pipeline
between STT/LLM/TTS so it observes without altering the frame stream. 5
unit tests pass (LatencyRecord e2e math + observer construction +
reset_turn). Full server suite: 18 passed.

---ci---
phase: 1
milestone: v0.1
plan: 02
task: 02-05,02-06
status: execute
persona: frontend-engineer,backend-engineer
requirements:
  covered: [REQ-VOICE-01, REQ-VOICE-02, REQ-VOICE-03, REQ-NFR-LAT-01]
---/ci---
2026-08-01 13:08:42 +00:00

192 lines
6.7 KiB
Python

"""Praxis Pipecat server pipeline — minimal viable voice loop (SLICE-02 TASK-02-04).
Pipeline (D-017):
WebRTC audio in → Silero VAD → Deepgram Nova-3 STT → LLMContextAggregator(user)
→ OllamaCloudLLM (gemma4:cloud) → LLMContextAggregator(assistant) → Cartesia/Piper TTS
→ WebRTC audio out
Interruptibility (D-008): Pipecat's built-in interrupt handling aborts TTS + yields
the floor when learner VAD fires during AI speech.
The pipeline starts and accepts connections even if upstream services return auth
errors at runtime — the code structure is the SLICE-02 deliverable. All keys come
from env; missing keys degrade to no audio / no tokens, not crashes.
Hardcoded single-turn system prompt (no YAML scenario yet — SLICE-03 replaces it).
"""
from __future__ import annotations
import os
from typing import Any
from loguru import logger
def _env(key: str, default: str = "") -> str:
return os.environ.get(key, default).strip()
# Hardcoded single-turn system prompt (SLICE-02 walking skeleton).
# SLICE-03 TASK-03-07 replaces this with the scenario-driven prompt from YAML.
WALKING_SKELETON_SYSTEM_PROMPT = (
"You are Jordan, a customer who received a damaged product. "
"You are frustrated but not abusive. You want a refund. "
"Stay in character. Do not break role. "
"Keep responses concise for voice (1-3 sentences)."
)
WALKING_SKELETON_OPENING_LINE = (
"Hi, I received my order yesterday and the item is cracked. I want my money back."
)
def _build_llm_context():
"""Build the LLMContext with the walking-skeleton system prompt."""
from pipecat.processors.aggregators.llm_context import LLMContext
messages = [
{"role": "system", "content": WALKING_SKELETON_SYSTEM_PROMPT},
]
return LLMContext(messages=messages)
def _build_transport(webrtc_connection) -> Any:
"""Build the SmallWebRTCTransport with audio in/out enabled."""
from pipecat.transports.base_transport import TransportParams
from pipecat.transports.smallwebrtc.transport import SmallWebRTCTransport
params = TransportParams(
audio_in_enabled=True,
audio_out_enabled=True,
audio_out_sample_rate=24000,
)
return SmallWebRTCTransport(webrtc_connection, params)
def _build_stt() -> Any:
"""Build the Deepgram Nova-3 STT service (D-013)."""
from pipecat.services.deepgram.stt import DeepgramSTTService
api_key = _env("DEEPGRAM_API_KEY")
if not api_key:
logger.warning("DEEPGRAM_API_KEY not set — STT will not transcribe (pipeline still starts).")
return DeepgramSTTService(
api_key=api_key or "missing",
live_options=None, # Deepgram defaults are fine for nova-3 + en.
)
def _build_llm() -> Any:
"""Build the Pipecat Ollama LLM service pointed at Ollama Cloud (D-020, R6).
Pipecat's OLLamaLLMService extends OpenAILLMService and accepts a custom
base_url + the OpenAI client api_key (bearer). We point it at
https://ollama.com/v1 with OLLAMA_API_KEY as the bearer.
"""
from pipecat.services.ollama.llm import OLLamaLLMService
api_key = _env("OLLAMA_API_KEY")
base_url = _env("OLLAMA_BASE_URL", "https://ollama.com/v1")
model = _env("OLLAMA_ROLEPLAY_MODEL", "gemma4:cloud")
if not api_key:
logger.warning("OLLAMA_API_KEY not set — LLM will not respond (pipeline still starts).")
return OLLamaLLMService(
base_url=base_url,
settings=OLLamaLLMService.Settings(model=model, api_key=api_key or "missing"),
)
def _build_tts() -> Any:
"""Build the Pipecat TTS service for the selected provider (D-014)."""
choice = _env("PRAXIS_TTS", "cartesia").lower()
if choice == "piper":
from pipecat.services.piper.tts import PiperTTSService
voice_model = _env("PIPER_VOICE_MODEL")
if not voice_model:
logger.warning("PIPER_VOICE_MODEL not set — Piper TTS will not speak (pipeline still starts).")
return PiperTTSService(
voice_id=voice_model or "missing",
)
# Default: Cartesia
from pipecat.services.cartesia.tts import CartesiaTTSService
api_key = _env("CARTESIA_API_KEY")
voice_id = _env("CARTESIA_VOICE_ID", "a3536a36-1d18-4efb-a95a-7c44b7b5e384")
if not api_key:
logger.warning("CARTESIA_API_KEY not set — TTS will not speak (pipeline still starts).")
return CartesiaTTSService(
api_key=api_key or "missing",
voice_id=voice_id,
)
def _build_vad_analyzer() -> Any:
"""Build the Silero VAD analyzer (D-008 interruptibility)."""
from pipecat.audio.vad.silero import SileroVADAnalyzer
return SileroVADAnalyzer()
def build_pipeline(webrtc_connection):
"""Assemble the full Pipecat pipeline + task + runner for one WebRTC session.
Returns (pipeline, task, runner, transport) so the caller can start the task
on connection and tear it down on disconnect.
"""
from pipecat.pipeline.pipeline import Pipeline
from pipecat.pipeline.runner import PipelineRunner
from pipecat.pipeline.task import PipelineParams, PipelineTask
from pipecat.processors.aggregators.llm_response_universal import (
LLMContextAggregator,
)
transport = _build_transport(webrtc_connection)
stt = _build_stt()
llm = _build_llm()
tts = _build_tts()
from server.latency import LatencyObserver
latency_observer = LatencyObserver()
context = _build_llm_context()
user_aggregator = LLMContextAggregator(context=context, role="user")
assistant_aggregator = LLMContextAggregator(context=context, role="assistant")
pipeline = Pipeline(
[
transport.input(), # WebRTC audio in
stt, # Deepgram Nova-3
latency_observer, # timestamp ASR-ready (TASK-02-06)
user_aggregator, # collect user transcript into context
llm, # Ollama gemma4:cloud
latency_observer, # timestamp LLM-first-token (passes through)
tts, # Cartesia/Piper
latency_observer, # timestamp TTS-first-audio + emit metric
transport.output(), # WebRTC audio out
assistant_aggregator, # collect assistant text into context
]
)
task = PipelineTask(
pipeline,
params=PipelineParams(
allow_interruptions=True, # D-008 abort-and-yield
enable_metrics=True, # latency measurement (TASK-02-06)
metrics_request_timeout=10.0,
),
)
runner = PipelineRunner(handle_sigint=False)
return pipeline, task, runner, transport
__all__ = [
"build_pipeline",
"WALKING_SKELETON_SYSTEM_PROMPT",
"WALKING_SKELETON_OPENING_LINE",
]