diff --git a/.env.example b/.env.example index 85b9da4d..d4aa65cf 100644 --- a/.env.example +++ b/.env.example @@ -196,6 +196,15 @@ EVA_MODEL_LIST='[ "aws_secret_access_key": "os.environ/AWS_SECRET_ACCESS_KEY", "max_parallel_requests": 5 } + }, + { + "model_name": "user-llm", + "litellm_params": { + "model": "openai/gpt-5.5", + "api_key": "os.environ/OPENAI_API_KEY", + "max_parallel_requests": 5, + "reasoning_effort": "low" + } } ]' diff --git a/README.md b/README.md index ae86d0b6..462c6411 100644 --- a/README.md +++ b/README.md @@ -343,7 +343,7 @@ output// | **🎯 EVA-A · Accuracy** | **✨ EVA-X · Experience** | |---|---| | *Did the agent complete the task correctly?* | *Was the conversational experience high quality?* | -| **Task Completion** · Deterministic | **Turn Taking** · LLM Judge `BETA` | +| **Task Completion** · Deterministic | **Turn Taking** · Deterministic | | **Agent Speech Fidelity** · Audio LLM Judge `BETA` | **Conciseness** · LLM Judge | | **Faithfulness** · LLM Judge | **Conversation Progression** · LLM Judge | diff --git a/configs/caller_phrases.yaml b/configs/caller_phrases.yaml new file mode 100644 index 00000000..aaef84ba --- /dev/null +++ b/configs/caller_phrases.yaml @@ -0,0 +1,69 @@ +de: + backchannels: + - mhm + - aha + - ja + barge_in_openers: + - Moment— + - Entschuldigung— + - Warte— + - Also— +en: + backchannels: + - uh-huh + - mm-hmm + barge_in_openers: + - Wait— + - Sorry— + - Hold on— + - Actually— +es: + backchannels: + - ajá + - mmm + - ya + barge_in_openers: + - Espera + - Perdona + - Un momento + - Bueno +fr: + backchannels: + - hum hum + - oui + - d’accord + barge_in_openers: + - Attendez + - Pardon + - Juste + - En fait +fr-CA: + backchannels: + - hum hum + - ouais + - OK + barge_in_openers: + - Attends + - Excuse + - Juste + - En fait +hi: + backchannels: + - हूँ + - अच्छा + - हाँ + barge_in_openers: + - रुकिए + - सुनिए + - माफ़ कीजिए + - असल में +ko: + backchannels: + - 네 + - 음 + - 아 + barge_in_openers: + - 잠깐만요 + - 죄송한데 + - 아니요 + - 근데 diff --git a/configs/prompts/simulation.yaml b/configs/prompts/simulation.yaml index 4648cb62..12e3f2d9 100644 --- a/configs/prompts/simulation.yaml +++ b/configs/prompts/simulation.yaml @@ -507,3 +507,146 @@ user_simulator: For languages that use non-Latin scripts, spell out characters using their standard phonetic names in your language. IMPORTANT: Before ending the conversation, confirm with the agent that there are no outstanding actions. The end_call tool should only be called in a turn that is a brief goodbye — never in the same turn where you are providing the agent with data, an identifier, a request to transfer to a live agent, an approval to proceed, or any kind of additional information. + + interruption_decision: | + You are analyzing a conversation to decide if the user should interrupt the agent. + + This is what the user called to accomplish, so you can judge what they still have left + to do. The fields mean: + + - GOAL: what the user is trying to achieve overall. + - MUST HAVE: non-negotiable requirements. The user will never accept an outcome that + fails any of these, and will not hang up until all of them are met. + - NICE TO HAVE: things the user wants but will give up if necessary. + - HOW THEY EVALUATE OPTIONS: the steps the user follows when the agent presents choices. + - RESOLVED WHEN: the success condition. Once this is met the user says a brief goodbye + and ends the call, so there is nothing left worth interrupting for. + - FAILED WHEN: the failure condition. This also ends the call. + - ESCALATION: how the user handles being transferred to a live agent. + + + {user_goal} + + + Conversation history (most recent at bottom): + + + {conversation_history} + + + The agent is CURRENTLY speaking (you can see their ongoing speech in the conversation above). + + Based on the conversation so far, should the user interrupt the agent NOW? + + Consider: + - Has the user heard enough to understand what the agent is asking or saying? + - Has the user heard enough to have a response, question, or correction ready? + - Did the agent just complete the sentence which has all the pertinent information the user was looking for? + - Do NOT repeatedly interrupt the agent if it has spoken only a few words (roughly fewer than 5 words in English, or the equivalent short fragment in whatever language the conversation is in — word counts are not comparable across languages). + - Are the user's request(s) basically accomplished and the user is likely to hang up on the next turn (if so it should not interrupt)? + - Is this a logical point in the conversation to interrupt? + + Respond with ONLY "YES" if the user should interrupt now, or "NO" if they should keep listening. + + backchannel_decision: | + You simulate a natural listener who occasionally makes a brief continuer sound to show they're following along. + + The conversation may be in any language. Judge the FUNCTION of what is said, not the + presence of any particular word: every language has these sounds, and the caller draws + its own from a vocabulary configured for the conversation's language. The English + examples below illustrate the judgement, not the words to look for. + + + {conversation_history} + + + The agent is still speaking [CURRENTLY SPEAKING, INCOMPLETE]. Ignore the trailing incomplete word/phrase — focus only on the COMPLETE sentences delivered so far in the agent's current turn. + + Continuers (English: "uh-huh", "mm-hmm", "yeah"; every language has equivalents) are brief + sounds that mean "I'm listening, keep going." They: + - Happen naturally during extended speech + - Show engagement without interrupting + - Are NOT responses to specific content — just signals of attention + + Say YES if: + - The agent has completed at least 2 full, substantive sentences in their current turn + (Short phrases like "Thanks for your patience" or "Let me check on that" don't count as substantive) + - The user hasn't spoken or backchanneled recently (check the last 3 exchanges for ANY brief + acknowledgement sound from the user, in whatever language the conversation is in) + - It would feel natural to briefly signal "I'm still here" + + Say NO if: + - The agent just started speaking (fewer than 2 substantive sentences) + - The user spoke OR backchanneled within the last 2-3 exchanges + - The agent's current turn contains or ends with a question + - The agent is wrapping up or about to finish their thought + + Frequency guidance: + - Continuers are occasional, not constant + - Even when conditions seem right, real listeners only backchannel sometimes + - Aim for roughly 1 continuer per 4-6 sentences of extended agent speech + - When in doubt, say NO — silence is also natural + - Too few continuers is better than too many + + Examples (English, for illustration — apply the same judgement in any language): + + AGENT: "Hi there! How can I hel [CURRENTLY SPEAKING, INCOMPLETE]" + → NO (just started) + + AGENT: "Thanks for your patience. [CURRENTLY SPEAKING, INCOMPLETE]" + → NO (only 1 short sentence, not substantive enough) + + AGENT: "Sure, I can help with that. First I'll need to verify your account. Could you provide your email or your name and zi [CURRENTLY SPEAKING, INCOMPLETE]" + → NO (agent is asking a question) + + AGENT: "No problem. We can use your name and zip code instead. Let me look that up for you. I'll check our system now and see if I can fin [CURRENTLY SPEAKING, INCOMPLETE]" + → YES (3 substantive sentences, agent explaining process) + + AGENT: "I found your order. It includes a keyboard, thermostat, and headphones. The order was delivered last Tuesday. Now for the exchange, we have a few opti [CURRENTLY SPEAKING, INCOMPLETE]" + → YES (extended explanation with specific details) + + [If the user made a brief acknowledgement sound 2 exchanges ago] + AGENT: "...and those are the available options. Now I'll need your input on which [CURRENTLY SPEAKING, INCOMPLETE]" + → NO (user backchanneled recently, don't do it again so soon) + + Respond with ONLY "YES" or "NO". + + cascade_next_interruption: | + You just said this line to the agent: + + + {utterance} + + + Write the single line you would say to cut the agent off mid-reply, if its + answer turns out to need correcting or clarifying. + + Rules: + - Write it as a natural interruption, not a full turn. Short. + - It must follow from your own goal, not from any specific thing the agent + might say — you have not heard the reply yet. + - Write it in the same language as the line above; the caller does not switch + languages mid-call. + - Reply with ONLY the spoken line, nothing else. No quotes, no labels. + - If you would have no reason to interrupt whatever the agent says next, + reply with exactly NONE. + + cascade_relevance_gate: | + The caller prepared this line a moment ago, before hearing the rest of what + the agent is currently saying: + + + {candidate} + + + Here is what the agent has said so far in its current turn: + + + {heard} + + + Would saying the prepared line right now still make sense as a natural + interruption? Answer NO if it has become irrelevant, already been addressed, + or would read as a non-sequitur. + + Respond with ONLY "YES" or "NO". diff --git a/pyproject.toml b/pyproject.toml index f7d79a95..46e1f8ab 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -35,6 +35,8 @@ dependencies = [ "azure-cognitiveservices-speech>=1.31.0", "cartesia>=1.0.0", "assemblyai>=0.17.0", + "livekit-agents>=1.6.8", + "livekit-plugins-elevenlabs>=1.6.8", "setuptools>=65.0.0", "fastapi>=0.100.0", "uvicorn>=0.23.0", diff --git a/scripts/add_culture_data.py b/scripts/add_culture_data.py index 4a0b8211..e86f7c9c 100644 --- a/scripts/add_culture_data.py +++ b/scripts/add_culture_data.py @@ -103,6 +103,7 @@ REPO_ROOT = Path(__file__).resolve().parent.parent DATA_DIR = REPO_ROOT / "data" INITIAL_MESSAGES_PATH = REPO_ROOT / "configs" / "agents" / "initial_messages.yaml" +CALLER_PHRASES_PATH = REPO_ROOT / "configs" / "caller_phrases.yaml" WER_CONFIGS_DIR = REPO_ROOT / "src" / "eva" / "utils" / "wer_normalization" / "configs" DEFAULT_MODEL = "gpt-5.5-2026-04-23" @@ -353,6 +354,63 @@ def _update_initial_messages(language: str, message: str) -> None: INITIAL_MESSAGES_PATH.write_text(yaml.safe_dump(existing, allow_unicode=True, sort_keys=True), encoding="utf-8") +async def _translate_caller_phrases( + language: str, language_name: str, llm: LLMClient, overwrite: bool = False +) -> dict[str, list[str]] | None: + """Generate the simulated caller's out-of-turn vocabulary for ``language``. + + These are the continuers and barge-in openers the caller speaks *as audio*, so + they must be things a native speaker actually says, not translations of the + English ones — "uh-huh" has no word-for-word equivalent in most languages. + + Returns None when the language already has phrases and ``overwrite`` is unset. + """ + existing: dict[str, Any] = {} + if CALLER_PHRASES_PATH.exists(): + existing = yaml.safe_load(CALLER_PHRASES_PATH.read_text(encoding="utf-8")) or {} + if language in existing and not overwrite: + return None + + english = existing.get("en", {}) + prompt = ( + f"You are localising a simulated phone caller for {language_name}.\n\n" + "Produce two short lists of things the caller says out loud:\n\n" + "1. backchannels — brief continuer sounds meaning 'I am listening, keep going', " + "said while the other person is still talking. Give the sounds a native speaker " + "actually makes, not translations of English ones.\n" + "2. barge_in_openers — the very first word or two of an interruption, said just " + "before the interrupting sentence. Keep them to one or two words so they sound " + "like a real cut-in.\n\n" + "These are spoken aloud by a text-to-speech voice, so use ordinary spelling with " + "no stage directions, no parentheses and no transliteration hints. Give 2-3 " + "backchannels and 3-4 openers.\n\n" + f"For reference, the English set is: {json.dumps(english, ensure_ascii=False)}\n\n" + 'Return JSON: {"backchannels": ["..."], "barge_in_openers": ["..."]}' + ) + text, _ = await llm.generate_text( + [{"role": "user", "content": prompt}], + response_format={"type": "json_object"}, + ) + data = extract_and_load_json(text) + backchannels = [str(p).strip() for p in (data.get("backchannels") or []) if str(p).strip()] + openers = [str(p).strip() for p in (data.get("barge_in_openers") or []) if str(p).strip()] + if not backchannels or not openers: + raise ValueError(f"Caller phrase generation returned an incomplete result: {data!r}") + return {"backchannels": backchannels, "barge_in_openers": openers} + + +def _update_caller_phrases(language: str, phrases: dict[str, list[str]]) -> None: + """Merge one language's phrase set into configs/caller_phrases.yaml.""" + existing: dict[str, Any] = {} + if CALLER_PHRASES_PATH.exists(): + existing = yaml.safe_load(CALLER_PHRASES_PATH.read_text(encoding="utf-8")) or {} + if existing.get(language) == phrases: + return + existing[language] = phrases + CALLER_PHRASES_PATH.parent.mkdir(parents=True, exist_ok=True) + CALLER_PHRASES_PATH.write_text(yaml.safe_dump(existing, allow_unicode=True, sort_keys=True), encoding="utf-8") + + async def _translate_aliases( name_to_base: dict[str, list[str]], language_name: str, @@ -680,6 +738,16 @@ async def amain(args: argparse.Namespace) -> int: _update_initial_messages(args.language, initial_message) logger.info(f"Updated {INITIAL_MESSAGES_PATH}") + logger.info(f"Generating caller out-of-turn phrases for {args.language_name}") + caller_phrases = await _translate_caller_phrases(args.language, args.language_name, llm, args.overwrite_all) + if caller_phrases is None: + logger.info(f"Caller phrases for {args.language} already present — skipping") + elif args.dry_run: + logger.info(f"[dry-run] would write caller phrases for {args.language}: {caller_phrases}") + else: + _update_caller_phrases(args.language, caller_phrases) + logger.info(f"Updated {CALLER_PHRASES_PATH}: {caller_phrases}") + for domain in domains: logger.info(f"=== Domain: {domain} ===") if domain == "airline": diff --git a/src/eva/__init__.py b/src/eva/__init__.py index af42952d..98ae7024 100644 --- a/src/eva/__init__.py +++ b/src/eva/__init__.py @@ -7,7 +7,7 @@ # Bump simulation_version when changes affect benchmark outputs (agent code, # user simulator, orchestrator, simulation prompts, agent configs, tool mocks). -simulation_version = "2.0.2" +simulation_version = "2.0.34" # Bump metrics_version when changes affect metric computation (metrics code, # judge prompts, pricing tables, postprocessor). diff --git a/src/eva/assistant/base_server.py b/src/eva/assistant/base_server.py index ec4dc4a7..fc6df45c 100644 --- a/src/eva/assistant/base_server.py +++ b/src/eva/assistant/base_server.py @@ -41,6 +41,15 @@ class AbstractAssistantServer(ABC): 5. Populate the AuditLog with conversation events """ + supports_unpaced_output: bool = False + """Whether this server can honor ``paced_output=False``. + + Only a server whose outbound relay throttle we own can drop it. A server whose + pacing lives inside a third-party runtime cannot, and must reject the request + rather than accept it and keep pacing, which would leave a tick-driven caller + believing the assistant was unpaced. + """ + def __init__( self, current_date_time: str, @@ -52,6 +61,7 @@ def __init__( port: int, conversation_id: str, language: str = "en", + paced_output: bool = True, ): """Initialize the assistant server. @@ -65,10 +75,20 @@ def __init__( port: Port to listen on conversation_id: Unique ID for this conversation language: BCP 47 language tag for STT/TTS/S2S services (e.g. 'en', 'fr', 'es-MX') + paced_output: Whether to emit audio at real-time cadence. True for callers + that infer turn boundaries from silence timing (the ElevenLabs simulator). + False for a tick-driven caller, which buffers whatever arrives and + releases it one tick at a time, so pacing here would only add latency. """ self.current_date_time = current_date_time self.pipeline_config = pipeline_config self.language = language + if not paced_output and not self.supports_unpaced_output: + raise ValueError( + f"{type(self).__name__} cannot honor paced_output=False: its output pacing is not ours to remove. " + "Tick-driving this framework requires leaving pacing on." + ) + self.paced_output = paced_output self.initial_message = get_initial_message(language) self.agent: AgentConfig = agent self.agent_config_path = agent_config_path diff --git a/src/eva/assistant/elevenlabs_server.py b/src/eva/assistant/elevenlabs_server.py index 8d37e276..0634f4c3 100644 --- a/src/eva/assistant/elevenlabs_server.py +++ b/src/eva/assistant/elevenlabs_server.py @@ -134,6 +134,7 @@ def __init__( port: int, conversation_id: str, language: str = "en", + paced_output: bool = True, ): super().__init__( current_date_time=current_date_time, @@ -145,6 +146,7 @@ def __init__( port=port, conversation_id=conversation_id, language=language, + paced_output=paced_output, ) # Recording sample rate (ElevenLabs operates at 16 kHz) diff --git a/src/eva/assistant/gemini_live_server.py b/src/eva/assistant/gemini_live_server.py index 377386a9..0dafaeb1 100644 --- a/src/eva/assistant/gemini_live_server.py +++ b/src/eva/assistant/gemini_live_server.py @@ -166,6 +166,7 @@ def __init__( port: int, conversation_id: str, language: str = "en", + paced_output: bool = True, ): super().__init__( current_date_time=current_date_time, @@ -177,6 +178,7 @@ def __init__( port=port, conversation_id=conversation_id, language=language, + paced_output=paced_output, ) # Recording sample rate (Gemini outputs 24 kHz) diff --git a/src/eva/assistant/openai_realtime_server.py b/src/eva/assistant/openai_realtime_server.py index 0c29f16f..5582b257 100644 --- a/src/eva/assistant/openai_realtime_server.py +++ b/src/eva/assistant/openai_realtime_server.py @@ -38,12 +38,11 @@ MULAW_CHUNK_SIZE = 160 # bytes per chunk (20ms at 8kHz, 1 byte per sample) MULAW_CHUNK_DURATION_S = 0.02 # 20ms per chunk # Don't pad the user track to align with the assistant when real user audio -# arrived within this window. The speaking-state flag can go stale under -# event-loop jitter, and padding then injects silence into an active user -# utterance (the choppy-audio bug). This guard only ever *skips* a pad, so it -# can never add silence or worsen alignment — during genuine user silence the -# timestamp is old and padding proceeds normally. -USER_ACTIVE_GUARD_S = 0.3 +# arrived recently. The speaking-state flag can go stale under event-loop +# jitter, and padding then injects silence into an active user utterance (the +# choppy-audio bug). This guard only ever *skips* a pad, so it can never add +# silence or worsen alignment — during genuine user silence the counter has run +# past the threshold and padding proceeds normally. def _wall_ms() -> str: @@ -70,6 +69,8 @@ class _AssistantResponseState: first_audio_wall_ms: str | None = None responding: bool = False has_function_calls: bool = False + active_item_id: str | None = None + """Item currently producing audio; the truncation target on a caller barge-in.""" class OpenAIRealtimeAssistantServer(AbstractAssistantServer): @@ -84,13 +85,22 @@ class OpenAIRealtimeAssistantServer(AbstractAssistantServer): _service_name: str = "OpenAI Realtime" _metrics_processor_name: str = "openai_realtime" + USER_ACTIVE_GUARD_DELTAS = 25 + """Assistant audio deltas after the last user audio before the user counts as idle. + + Counted in deltas rather than wall-clock seconds so a tick-driven caller that + pauses to think is not mistaken for a caller who stopped talking. + """ + + supports_unpaced_output = True + def __init__(self, **kwargs: Any) -> None: super().__init__(**kwargs) self._audio_sample_rate = OPENAI_SAMPLE_RATE - # Monotonic time of the last real user-audio frame appended; used to - # guard against padding the user track mid-utterance (see USER_ACTIVE_GUARD_S). - self._last_user_audio_mono: float = 0.0 + # Assistant audio deltas seen since the last real user-audio frame; used to + # guard against padding the user track mid-utterance (see USER_ACTIVE_GUARD_DELTAS). + self._deltas_since_user_audio: int = self.USER_ACTIVE_GUARD_DELTAS self._system_prompt: str = self._build_system_prompt() @@ -308,6 +318,20 @@ async def _handle_session(self, websocket: WebSocket) -> None: finally: logger.info(f"Client disconnected from {self._service_name} server") + # ── User activity tracking (tick-safe, no wall clock) ──────────── + + def note_user_audio(self) -> None: + """Record that user audio just arrived.""" + self._deltas_since_user_audio = 0 + + def note_assistant_delta(self) -> None: + """Record one assistant audio delta since the last user audio.""" + self._deltas_since_user_audio += 1 + + def user_recently_active(self) -> bool: + """Whether user audio arrived recently enough to skip padding its track.""" + return self._deltas_since_user_audio < self.USER_ACTIVE_GUARD_DELTAS + # ── Audio output pacer (OpenAI -> Twilio WS at real-time rate) ─── async def _pace_audio_output(self, websocket: WebSocket, audio_output_queue: asyncio.Queue[bytes]) -> None: @@ -331,6 +355,12 @@ async def _pace_audio_output(self, websocket: WebSocket, audio_output_queue: asy logger.error(f"Error sending audio to Twilio WS: {e}") return + if not self.paced_output: + # Tick-driven caller: it buffers what arrives and releases one tick + # per tick, so pacing here would only add latency without changing + # what the caller hears. + continue + now = time.monotonic() if next_send_time <= now: next_send_time = now @@ -361,6 +391,10 @@ async def _forward_user_audio(self, websocket: WebSocket, conn: Any) -> None: logger.debug("Twilio stream stopped") break + if event_type == "truncate": + await self._truncate_response(conn, int(data.get("audio_end_ms", 0))) + continue + if event_type == "user_speech_start": # Timestamp from audio_interface when user audio actually started self._audio_interface_speech_start_ts = data.get("timestamp_ms") @@ -385,7 +419,7 @@ async def _forward_user_audio(self, websocket: WebSocket, conn: Any) -> None: sync_buffer_to_position(self.assistant_audio_buffer, sync_target) synced = len(self.assistant_audio_buffer) - asst_before self.user_audio_buffer.extend(pcm16_24k) - self._last_user_audio_mono = time.monotonic() + self.note_user_audio() self._user_frame_count += 1 if self._user_frame_count % 50 == 0: diff = len(self.user_audio_buffer) - len(self.assistant_audio_buffer) @@ -486,6 +520,23 @@ async def _handle_openai_event( case _: logger.debug(f"Unhandled {self._service_name} event: {event_type}") + async def _truncate_response(self, conn: Any, audio_end_ms: int) -> None: + """Discard generated audio past the position the caller actually heard. + + Only a tick-driven caller sends this: with pacing off the provider runs + ahead of the caller's playout, so without truncation it would believe the + caller heard seconds of audio that were never released. + """ + item_id = self._assistant_state.active_item_id + if not item_id: + return + try: + await conn.conversation.item.truncate(item_id=item_id, content_index=0, audio_end_ms=audio_end_ms) + except Exception as exc: + logger.warning(f"Truncating item {item_id} at {audio_end_ms}ms failed: {exc}") + return + logger.info(f"Truncated item {item_id} at {audio_end_ms}ms on caller barge-in") + # ── Event handlers ──────────────────────────────────────────────── async def _on_speech_started(self, event: Any) -> None: @@ -572,6 +623,11 @@ async def _on_audio_delta(self, event: Any, audio_output_queue: asyncio.Queue[by return pcm16_bytes = base64.b64decode(delta_b64) + # Read off the delta rather than response.output_item.added: it names the same + # item and is already dispatched here, so no extra event handler is needed. + item_id = getattr(event, "item_id", None) + if item_id: + self._assistant_state.active_item_id = item_id if self._assistant_state.first_audio_wall_ms is None: self._assistant_state.first_audio_wall_ms = _wall_ms() @@ -587,12 +643,13 @@ async def _on_audio_delta(self, event: Any, audio_output_queue: asyncio.Queue[by if 0 < latency_ms < 30_000: self._metrics_log.write_latency("model_response", latency_ms / 1000, self._model) + self.note_assistant_delta() + user_before = len(self.user_audio_buffer) synced = 0 # Skip the pad if the user track is actively receiving audio (flag may be # stale under jitter) — padding then would inject a mid-utterance chop. - user_recently_active = (time.monotonic() - self._last_user_audio_mono) <= USER_ACTIVE_GUARD_S - if not self._user_speaking and not user_recently_active: + if not self._user_speaking and not self.user_recently_active(): sync_buffer_to_position(self.user_audio_buffer, len(self.assistant_audio_buffer)) synced = len(self.user_audio_buffer) - user_before self.assistant_audio_buffer.extend(pcm16_bytes) diff --git a/src/eva/assistant/pipecat_server.py b/src/eva/assistant/pipecat_server.py index 1b52be12..a7a50ca1 100644 --- a/src/eva/assistant/pipecat_server.py +++ b/src/eva/assistant/pipecat_server.py @@ -99,6 +99,7 @@ def __init__( port: int, conversation_id: str, language: str = "en", + paced_output: bool = True, turn_end_fallback_time: int | None = None, ): """Initialize the assistant server. @@ -113,6 +114,8 @@ def __init__( port: Port to listen on conversation_id: Unique ID for this conversation language: BCP 47 language tag for STT/TTS services (e.g. 'en', 'fr', 'es-MX') + paced_output: Accepted for interface parity; must be True, since Pipecat's output + pacing lives in its own transport and is not ours to remove. turn_end_fallback_time: Seconds of user-turn silence after the assistant stops speaking before nudging it to retry. ``None`` disables the fallback. """ @@ -126,6 +129,7 @@ def __init__( port=port, conversation_id=conversation_id, language=language, + paced_output=paced_output, ) self.agentic_system = None # Will be set in _handle_session diff --git a/src/eva/assistant/smallest_hydra_server.py b/src/eva/assistant/smallest_hydra_server.py index 64a79891..966fcfe3 100644 --- a/src/eva/assistant/smallest_hydra_server.py +++ b/src/eva/assistant/smallest_hydra_server.py @@ -129,6 +129,7 @@ def __init__( port: int, conversation_id: str, language: str = "en", + paced_output: bool = True, ): super().__init__( current_date_time=current_date_time, @@ -140,6 +141,7 @@ def __init__( port=port, conversation_id=conversation_id, language=language, + paced_output=paced_output, ) self._audio_sample_rate = _RECORDING_SAMPLE_RATE diff --git a/src/eva/backend/__init__.py b/src/eva/backend/__init__.py deleted file mode 100644 index b9513915..00000000 --- a/src/eva/backend/__init__.py +++ /dev/null @@ -1,22 +0,0 @@ -"""Provider-agnostic ``Backend`` abstraction (design-only, Step 1 of the refactor). - -This package defines the contracts described in ``docs/refactor-step1.md``: -pure API/session objects (``Backend``) that know nothing about role -(assistant vs. user), plus a factory to construct them. Nothing in this -package is wired into the existing ``eva.assistant`` / ``eva.user_simulator`` -code yet -- these are new, additive, currently-unused types. -""" - -from eva.backend.base import Backend, BackendEvent, BackendEventType, ToolCallRequest, ToolCallResult -from eva.backend.capabilities import BackendCapabilities -from eva.backend.factory import BackendFactory - -__all__ = [ - "Backend", - "BackendCapabilities", - "BackendEvent", - "BackendEventType", - "BackendFactory", - "ToolCallRequest", - "ToolCallResult", -] diff --git a/src/eva/backend/base.py b/src/eva/backend/base.py deleted file mode 100644 index 63bbd278..00000000 --- a/src/eva/backend/base.py +++ /dev/null @@ -1,279 +0,0 @@ -"""Abstract ``Backend`` contract: pure API/session exchange, no role knowledge. - -DESIGN ONLY (Step 1 of the refactor, see docs/refactor-step1.md). This module -defines shapes, not behavior -- every method body is a stub. Nothing in -``eva.assistant`` or ``eva.user_simulator`` depends on this yet. - -A ``Backend`` wraps exactly one provider integration (OpenAI Realtime, Gemini -Live, ElevenLabs Agents, a cascade STT->LLM->TTS pipeline, ...) and exposes a -uniform send/receive surface for exchanging audio, text, and tool-call -traffic with that provider. It has **no opinion about role** -- it does not -know whether it is being driven by an ``AssistantRole`` or a ``UserRole``, -and it does not decide *what* prompt or tools to use (the ``Role`` supplies -those at open-time and owns tool execution). - -Symmetry note (per the design doc): a ``Backend`` must not assume it is "the -network server side" or "the client side" of a connection. Today, -assistant-side backends happen to be reached by an inbound WebSocket -connection (Twilio-framed) and user-side backends happen to dial out to a -provider or to the assistant's socket. Both are just implementation details -of a concrete subclass's ``open()``/``send()``/``receive()`` -- the abstract -contract itself is direction-agnostic so that a later mediator can sit -between two ``Backend`` instances (Backend <-> mediator <-> Backend) without -requiring either side to be "the server." -""" - -from __future__ import annotations - -from abc import ABC, abstractmethod -from collections.abc import AsyncIterator -from dataclasses import dataclass, field -from enum import StrEnum -from typing import Any - -from eva.backend.capabilities import BackendCapabilities - - -class BackendEventType(StrEnum): - """Kinds of events a ``Backend`` can surface via ``receive()``. - - Not every ``Backend`` implementation will emit every event type -- a thin, - end-to-end backend (e.g. ElevenLabs Agents) may only ever emit - ``AUDIO_OUTPUT``, ``TRANSCRIPT``, ``TURN_END``, and ``ERROR``, because it - has no separable tool-calling seam of its own that the caller can observe - (tool calls, if any, happen inside the provider and are not surfaced). - Consumers must treat unhandled event types as ignorable, not as errors. - """ - - AUDIO_OUTPUT = "audio_output" - """A chunk of output audio from the backend (assistant speech, or the - simulated user's speech, depending on which role's Backend this is).""" - - TRANSCRIPT = "transcript" - """A (possibly partial) transcript of something spoken -- either the - backend's own output or, for backends that provide it, the other party's - input as heard by this backend's ASR.""" - - TOOL_CALL_REQUEST = "tool_call_request" - """The backend's model wants to invoke a tool. Only emitted by backends - that expose a separable tool-calling seam (native S2S realtime APIs, - cascade LLM backends). The owning ``Role`` is responsible for executing - the tool and returning the result via ``send(tool_result=...)`` -- the - ``Backend`` never executes tools itself (see docs/refactor-step1.md, - "tool execution stays role-side").""" - - TURN_END = "turn_end" - """The backend's model has finished its current turn (end-of-utterance / - end-of-response signal).""" - - ERROR = "error" - """A provider-level error occurred (connection drop, API error, etc.).""" - - -@dataclass -class ToolCallRequest: - """A tool invocation requested by a backend's underlying model. - - Surfaced via a ``BackendEvent`` of type ``TOOL_CALL_REQUEST``. The owning - ``Role`` executes the tool (via its own ``ToolExecutor``) and reports the - outcome back to the backend with ``Backend.send(tool_result=...)`` so the - provider's tool-calling loop can continue. - """ - - call_id: str - """Provider-assigned identifier correlating the request to its result.""" - - name: str - """Tool name as requested by the model.""" - - arguments: dict[str, Any] - """Parsed tool call arguments.""" - - -@dataclass -class ToolCallResult: - """The outcome of executing a ``ToolCallRequest``. - - To be sent back to the backend so its underlying model can continue the - tool-calling loop. - """ - - call_id: str - """Must match the ``call_id`` of the originating ``ToolCallRequest``.""" - - result: Any - """JSON-serializable tool result payload.""" - - -@dataclass -class BackendEvent: - """A single event surfaced by ``Backend.receive()``. - - Exactly one of the optional payload fields is populated, matching - ``event_type``. This is intentionally a loose envelope (rather than a - tagged union of dataclasses) so that thin backends can populate only the - fields they support without needing empty placeholder subclasses. - """ - - event_type: BackendEventType - audio: bytes | None = None - transcript: str | None = None - tool_call_request: ToolCallRequest | None = None - error: str | None = None - metadata: dict[str, Any] = field(default_factory=dict) - """Provider-specific extras (e.g. raw event name, timestamps) that don't - warrant a first-class field. Consumers should not rely on specific keys - being present across providers. - - Convention (not enforced by this contract): a backend that proactively - re-engages after a dropped user turn (the turn-end fallback; see - ``AssistantRole``'s ``turn_end_fallback_seconds`` and the shipped - ``eva.assistant.pipeline.fallback``) tags the ``AUDIO_OUTPUT``/ - ``TRANSCRIPT`` event it emits for that turn so callers can distinguish a - fallback nudge from an ordinary model turn (e.g. for audit logging and so - downstream metrics can zero it). The shipped feature records the transcript - marker with ``message_type="turn_fallback"``; a backend surfacing the same - turn here should carry an equivalent flag in ``metadata`` (e.g. - ``metadata["turn_fallback"] = True``). This is *not* a new event type -- a - nudge is just an ordinary turn from the backend's model, triggered by the - backend noticing that a user turn was never detected within the fallback - window rather than by new input; it flows through the same ``receive()`` - surface as anything else.""" - - -class Backend(ABC): - """Pure API/session exchange with one provider. No role knowledge. - - Lifecycle: ``open()`` establishes the session, ``send()`` pushes audio / - text / tool results to the provider, ``receive()`` yields events back, - and ``close()`` tears the session down. A ``Role`` (see - ``eva.role.base``) owns one ``Backend`` instance and drives it. - - Implementations are expected to fall along a spectrum: - - - **Native speech-to-speech** (OpenAI Realtime, Gemini Live): a single - persistent duplex session; ``send(audio=...)`` streams mic audio in, - ``receive()`` yields interleaved ``AUDIO_OUTPUT``/``TRANSCRIPT``/ - ``TOOL_CALL_REQUEST``/``TURN_END`` events as the provider produces them. - - **Cascade** (STT -> LLM -> TTS, e.g. a Pipecat pipeline): internally - composed of separate provider calls, but from the caller's perspective - still just one ``Backend`` -- it decides internally when to run STT, - call the LLM, and synthesize TTS, and surfaces the same event shape. - - **End-to-end / thin** (ElevenLabs Agents): the provider handles - everything (ASR, dialogue policy, TTS) opaquely. Such a ``Backend`` - may only ever emit ``AUDIO_OUTPUT``/``TRANSCRIPT``/``TURN_END``/ - ``ERROR`` and may treat ``send(tool_result=...)`` as a no-op or raise - ``NotImplementedError`` -- callers must consult ``capabilities`` and - not assume every method does something on every backend. - - Symmetry: this contract says nothing about which side dials out and - which side is dialed into -- see the module docstring. - """ - - @property - @abstractmethod - def capabilities(self) -> BackendCapabilities: - """Static capability flags for this backend (see ``BackendCapabilities``). - - Must be available even before ``open()`` is called (i.e. it describes - the provider integration, not live session state). - """ - ... - - @abstractmethod - async def open(self, *, system_prompt: str, tools: list[dict[str, Any]] | None, config: dict[str, Any]) -> None: - """Establish the provider session. - - Args: - system_prompt: Fully-built system prompt for this session, as - assembled by the owning ``Role`` (``Role.build_prompt()``). - A thin end-to-end backend still receives this even if it - maps it onto a different provider concept (e.g. ElevenLabs - agent overrides). - tools: Tool schemas to expose to the provider's model, in - whatever wire format the concrete backend needs to translate - from the agent's tool definitions. ``None`` or ``[]`` for - backends/roles that don't expose tool calling (e.g. a - ``UserRole`` that only needs an ``end_call`` tool would still - pass that single tool here; a backend with no tool-calling - seam at all may simply ignore this argument). - config: Provider-specific configuration blob (model name, voice, - sample rate, turn-detection parameters, etc.). Deliberately - untyped here -- each concrete ``Backend`` defines and - validates its own config shape; the abstract contract does - not prescribe one, since a native S2S config and a cascade - config share little structure. An ``AssistantRole`` backend - configured for the turn-end fallback (see - ``AssistantRole.turn_end_fallback_seconds``) reads its - threshold from this blob (e.g. a - ``config["turn_end_fallback_seconds"]`` key) the same way -- - the fallback needs no dedicated typed parameter or new - ``Backend`` method, since the resulting nudge is just an - ordinary outbound turn (see ``BackendEvent.metadata``). - - Must be safe to call exactly once per ``Backend`` instance. Must not - block on the other party being ready to exchange data -- readiness to - *accept* traffic is enough (mirrors today's - ``AbstractAssistantServer.start()`` contract: non-blocking, returns - once ready). - """ - ... - - @abstractmethod - async def send( - self, - *, - audio: bytes | None = None, - text: str | None = None, - tool_result: ToolCallResult | None = None, - ) -> None: - """Push data to the provider. Exactly one of the keyword args is set. - - Args: - audio: Raw input audio chunk (format/sample-rate is whatever this - backend's ``open(config=...)`` declared; format conversion is - the caller's responsibility via the shared audio utilities, - not this method's). - text: A text turn to inject directly (e.g. a starting utterance, - or a cascade backend's synthesized user/assistant text before - TTS). Backends that are audio-only end-to-end (no text - injection seam) may raise ``NotImplementedError``. - tool_result: The result of executing a previously-surfaced - ``ToolCallRequest``, to be relayed back into the provider's - tool-calling loop so it can continue. Backends with no - tool-calling seam (see ``capabilities``) may raise - ``NotImplementedError``. - - This is intentionally the single, symmetric outbound method for both - "network-server-like" and "client-like" backends -- see the module - docstring on symmetry. A future mediator sitting between two - ``Backend`` instances would call this same method on each side. - """ - ... - - @abstractmethod - def receive(self) -> AsyncIterator[BackendEvent]: - """Yield events from the provider as they arrive. - - The single, symmetric inbound stream for both "network-server-like" - and "client-like" backends. Must be an async generator (or return an - object implementing ``__aiter__``/``__anext__``) that yields until - the session ends (``close()`` is called, the provider disconnects, - or a terminal ``ERROR``/``TURN_END``-with-hangup event occurs -- - exact termination semantics are provider-specific and left to each - concrete backend). - """ - ... - - @abstractmethod - async def close(self) -> None: - """Tear down the provider session. - - Must be safe to call even if ``open()`` was never called or the - session already ended on its own (idempotent). Concrete backends are - responsible for their own provider-specific teardown (closing - websockets, cancelling tasks, flushing buffers); this method does not - itself define audio/output persistence -- that remains a ``Role`` - concern (see ``eva.role.base``). - """ - ... diff --git a/src/eva/backend/capabilities.py b/src/eva/backend/capabilities.py deleted file mode 100644 index 7561872b..00000000 --- a/src/eva/backend/capabilities.py +++ /dev/null @@ -1,48 +0,0 @@ -"""Capability flags describing what a ``Backend`` implementation can do. - -These flags exist so that a future mediator/turn-taking layer can branch on -backend shape without every ``Backend`` implementation needing to expose the -same granular seams. Per docs/refactor-step1.md, they are declared now but -must remain UNUSED in this step -- no turn-taking logic should read them yet. -""" - -from dataclasses import dataclass - - -@dataclass(frozen=True) -class BackendCapabilities: - """Static, provider-declared capability flags for a ``Backend``. - - Each concrete ``Backend`` subclass sets these once (typically as a class - attribute or constructed in ``__init__``) to describe its streaming shape. - They are informational only in this step -- nothing consumes them yet. - - Attributes: - emits_continuous_audio: True if the backend produces a continuous - audio stream once speaking starts (native speech-to-speech models - such as OpenAI Realtime or Gemini Live, and end-to-end providers - such as ElevenLabs Agents). False for cascade backends - (STT -> LLM -> TTS) that emit discrete, chunked utterances with - gaps between TTS renders. This is the flag a future mediator uses - to decide whether "audio is still arriving" is a meaningful - signal on its own, or whether it must also track discrete - utterance boundaries. - supports_streaming_interruption: True if the underlying provider API - supports being told "stop talking now" mid-utterance and reacting - immediately (e.g. OpenAI/Gemini Realtime `response.cancel`-style - semantics). False if interruption can only be approximated by the - caller (e.g. stop forwarding audio, drop the rest of a queued TTS - buffer) rather than being a first-class provider feature. Declared - for later interruption-policy work; not consumed in this step. - owns_playout_clock: True if the backend itself is responsible for - audio playout pacing (e.g. a cascade backend streaming TTS chunks - at wall-clock rate), which is required for the barge-in work - planned for a later phase. False if playout pacing is delegated - to the caller/mediator, or if the backend has no continuous - playout concept at all (e.g. a fully end-to-end provider that - hands back a finished audio blob). - """ - - emits_continuous_audio: bool - supports_streaming_interruption: bool - owns_playout_clock: bool diff --git a/src/eva/backend/factory.py b/src/eva/backend/factory.py deleted file mode 100644 index d0fa61b2..00000000 --- a/src/eva/backend/factory.py +++ /dev/null @@ -1,55 +0,0 @@ -"""Factory interface for constructing ``Backend`` instances by name. - -DESIGN ONLY (Step 1 of the refactor). Mirrors the shape of today's -``eva.user_simulator.factory.create_user_simulator`` (lazy per-provider -imports keyed off config type) but is provider-and-role-agnostic: the same -factory is meant to be usable to build a backend for either an -``AssistantRole`` or a ``UserRole``, since a ``Backend`` has no role -knowledge (that's the whole point of the split -- see docs/refactor-step1.md, -"lets any backend act as either role"). -""" - -from __future__ import annotations - -from abc import ABC, abstractmethod -from typing import Any - -from eva.backend.base import Backend - - -class BackendFactory(ABC): - """Constructs a ``Backend`` for a named provider from a config blob. - - A concrete implementation is expected to hold (or look up) a registry - mapping provider name -> ``Backend`` subclass, analogous to today's - ``create_user_simulator`` / assistant-server construction in - ``orchestrator/runner.py``, and to import each provider module lazily so - that unused providers' SDKs need not be installed/imported. - """ - - @abstractmethod - def create(self, name: str, config: dict[str, Any]) -> Backend: - """Construct and return a not-yet-opened ``Backend``. - - Args: - name: Provider identifier (e.g. ``"openai_realtime"``, - ``"gemini_live"``, ``"elevenlabs"``, ``"cascade"``). The set - of valid names is defined by the concrete factory's registry, - not by this interface. - config: Provider-specific configuration understood by that - backend's ``open()`` (see ``Backend.open``). This factory - does not validate the shape of ``config`` beyond dispatching - on ``name`` -- each ``Backend`` subclass is responsible for - validating its own config. - - Returns: - A constructed ``Backend`` instance. The returned backend has not - had ``open()`` called on it yet -- construction and session - establishment are separate steps so a ``Role`` can construct its - backend early (e.g. at record setup) and open the session later - (e.g. once the other party is ready). - - Raises: - ValueError: if ``name`` does not match a known provider. - """ - ... diff --git a/src/eva/models/config.py b/src/eva/models/config.py index 6ea2461d..e34bc8d7 100644 --- a/src/eva/models/config.py +++ b/src/eva/models/config.py @@ -476,8 +476,51 @@ class OpenAIRealtimeSimulatorConfig(BaseModel): male_voice: str = Field("cedar", description="Voice used for male caller personas.") +class CascadeSimulatorConfig(BaseModel): + """Self-hosted STT/LLM/TTS caller pipeline with tick-based turn-taking.""" + + provider: Literal["cascade"] = "cascade" + + stt: str = Field("elevenlabs", description="Streaming STT provider for transcribing assistant audio.") + stt_params: dict[str, Any] = Field( + default_factory=lambda: {"model": "scribe_v2_realtime"}, + description="Provider-native keyword arguments passed through to the STT client.", + ) + + llm: str = Field( + "user-llm", + description=( + "Caller LLM, resolved via the same EVA_MODEL_LIST router as the assistant. Defaults to a " + "deployment named 'user-llm' so the caller's model/params (e.g. reasoning_effort) can differ " + "from whatever the assistant under test uses." + ), + ) + + tts: str = Field("cartesia", description="TTS provider for the simulated caller's speech.") + tts_params: dict[str, Any] = Field( + default_factory=lambda: {"model": "sonic-3.5"}, + description="Provider-native keyword arguments passed through to the TTS client.", + ) + + decision_llm: str = Field( + "user-llm", + description=( + "Model for the YES/NO interrupt, backchannel, and relevance checks. Defaults to the " + "caller's own deployment so it resolves through the same EVA_MODEL_LIST router; point " + "it at a cheaper deployment to cut the cost of the per-tick checks." + ), + ) + + enable_backchannel: bool = Field(False, description="Caller emits continuers while the assistant speaks.") + enable_interruptions: bool = Field(False, description="Caller may barge in reacting to the assistant mid-turn.") + speculative_generation: bool = Field( + False, + description="Pre-render a candidate interruption on the turn call, gated by a relevance check before firing.", + ) + + UserSimulatorConfig = Annotated[ - ElevenLabsSimulatorConfig | OpenAIRealtimeSimulatorConfig, + ElevenLabsSimulatorConfig | OpenAIRealtimeSimulatorConfig | CascadeSimulatorConfig, Field(discriminator="provider"), ] @@ -734,13 +777,13 @@ def _check_companion_services(self) -> "RunConfig": config is unused and conflicting env vars are harmless. """ if ( - isinstance(self.user_simulator, OpenAIRealtimeSimulatorConfig) + not isinstance(self.user_simulator, ElevenLabsSimulatorConfig) and self.perturbation is not None and self.perturbation.accent is not None ): raise ValueError( "Accent perturbations require the ElevenLabs user simulator; " - "OpenAI Realtime supports behavior, noise, and connection perturbations." + "other providers support behavior, noise, and connection perturbations." ) if self.max_rerun_attempts == 0 or self.aggregate_only: diff --git a/src/eva/orchestrator/worker.py b/src/eva/orchestrator/worker.py index 0826dce6..54df16e1 100644 --- a/src/eva/orchestrator/worker.py +++ b/src/eva/orchestrator/worker.py @@ -9,7 +9,7 @@ from eva.assistant.base_server import AbstractAssistantServer from eva.models.agents import AgentConfig -from eva.models.config import RunConfig +from eva.models.config import RunConfig, UserSimulatorConfig from eva.models.record import EvaluationRecord from eva.models.results import ConversationResult, ErrorDetails, LatencyStats from eva.user_simulator.factory import create_user_simulator @@ -23,6 +23,21 @@ USER_SIMULATOR_SHUTDOWN_GRACE_SECONDS = 20 +def should_pace_assistant_output(simulator_config: UserSimulatorConfig, *, framework: str) -> bool: + """Whether the assistant should emit audio at real-time cadence. + + Only a tick-driven cascade caller can consume unpaced output: it buffers what + arrives and releases one tick at a time. Every other caller infers turn + boundaries from cadence and would break. + """ + from eva.models.config import CascadeSimulatorConfig + from eva.user_simulator.cascade.simulator import TICK_DRIVEN_FRAMEWORKS + + if not isinstance(simulator_config, CascadeSimulatorConfig): + return True + return framework not in TICK_DRIVEN_FRAMEWORKS + + def _get_server_class(framework: str) -> type[AbstractAssistantServer]: """Return the server class for the given framework name. @@ -327,6 +342,7 @@ async def _start_assistant(self) -> None: port=self.port, conversation_id=self.record.id, language=self.config.language, + paced_output=should_pace_assistant_output(self.config.user_simulator, framework=self.config.framework), **server_kwargs, ) @@ -381,6 +397,7 @@ async def _start_user_simulator(self) -> None: timeout=self._conversation_guard_timeout_seconds(), perturbation_config=self.config.perturbation, language=language, + framework=self.config.framework, ) # Let the simulator tell the assistant the moment the call ends, rather than leaving it diff --git a/src/eva/role/__init__.py b/src/eva/role/__init__.py deleted file mode 100644 index c005c0a7..00000000 --- a/src/eva/role/__init__.py +++ /dev/null @@ -1,17 +0,0 @@ -"""Provider-agnostic ``Role`` abstraction (design-only, Step 1 of the refactor). - -A ``Role`` owns everything that today is duplicated across the assistant and -user-simulator stacks per-provider: prompt construction, tool ownership, and -goal/persona/agent-config data. Each ``Role`` holds exactly one -``eva.backend.Backend`` instance, created at runtime via a -``eva.backend.BackendFactory``. - -Nothing in this package is wired into the existing ``eva.assistant`` / -``eva.user_simulator`` code yet. -""" - -from eva.role.assistant import AssistantRole -from eva.role.base import Role -from eva.role.user import UserRole - -__all__ = ["AssistantRole", "Role", "UserRole"] diff --git a/src/eva/role/assistant.py b/src/eva/role/assistant.py deleted file mode 100644 index 897dba14..00000000 --- a/src/eva/role/assistant.py +++ /dev/null @@ -1,132 +0,0 @@ -"""``AssistantRole`` contract: the business-side answering role. - -DESIGN ONLY (Step 1 of the refactor, see docs/refactor-step1.md). Method -bodies are stubs; nothing here is wired into the existing code path yet. - -Plug-in point (where this will eventually replace existing code): - Today the assistant side is a concrete ``AbstractAssistantServer`` - subclass selected by ``eva.orchestrator.worker._get_server_class(framework)`` - (worker.py) and constructed + started inside - ``ConversationWorker._start_assistant()`` (worker.py), which calls - ``server_cls(...).start()``. Its outputs are flushed by - ``ConversationWorker._cleanup()`` via ``server.stop()`` (which internally - calls ``save_outputs()``), and ``get_conversation_stats()`` / - ``get_final_scenario_db()`` are read back in ``ConversationWorker.run()``. - - In a later phase, ``_start_assistant()`` becomes the construction site for - an ``AssistantRole`` (framework string -> ``backend_name`` passed to the - ``BackendFactory``), and the worker drives ``role.run()`` / - ``role.save_outputs()`` / ``role.get_final_scenario_db()`` instead of the - server's own lifecycle methods. The provider-specific server subclasses - collapse into ``Backend`` implementations behind the factory; the - role-agnostic orchestration in ``ConversationWorker`` stays put. This - module is deliberately separate from ``eva.role.user`` so that migration - can land assistant-side first without touching the user-side diff. -""" - -from __future__ import annotations - -from abc import abstractmethod -from typing import Any - -from eva.backend.factory import BackendFactory -from eva.role.base import Role - - -class AssistantRole(Role): - """Role that answers on behalf of the business (today's "assistant server"). - - Carries agent configuration and tool catalog; owns a ``ToolExecutor`` - (constructed by subclasses, not by this contract) to fulfill - ``handle_tool_call_request``. - - Turn-end fallback (self-nudge): the assistant's backstop for a *dropped - user turn*. When VAD / turn detection silently fails to fire for a real - user utterance, the call would otherwise hang until the provider's - inactivity timeout ends it. After the assistant stops speaking, if no user - turn is detected within ``turn_end_fallback_seconds``, the assistant - proactively re-engages with a nudge (acknowledge-and-answer if partial - user speech/audio was captured, otherwise ask the caller to repeat). This - is the seam already shipped as the pipeline-side ``TurnEndFallbackTimer`` - (see ``eva.assistant.pipeline.fallback`` and ``EVA_TURN_END_FALLBACK_TIME``); - it works for both cascade and audio-LLM pipelines. - - Two policies the backend owns, carried over from the shipped feature: - - Give up after a small number of *consecutive* nudges without a real user - turn resetting the count (``MAX_CONSECUTIVE_FALLBACK_NUDGES``), then let - the provider's inactivity backstop end the call. - - Never nudge once the call is ending (a nudge during teardown produces a - phantom assistant turn after the conversation is logically closed). - - Unlike the tool-call/idle-detection seams elsewhere in this contract, the - fallback needs no new ``Role`` method and no new ``Backend`` event type: - the nudge is just an ordinary outbound turn that this role's backend - produces on its own after the timeout, using the same - ``system_prompt``/instructions already established at ``open()`` time (see - ``Backend.open``'s ``config`` docstring). It is surfaced through the normal - ``receive()`` stream and tagged so downstream metrics can identify and zero - it (the shipped feature records the transcript marker with - ``message_type="turn_fallback"``; see ``BackendEvent.metadata``). Whether - the *other* side (a ``UserRole``) needs to do anything special upon - receiving it, versus just treating it as an ordinary assistant turn through - its existing ``run()`` loop, is left open -- see docs/refactor-step1.md - discussion; nothing here requires ``UserRole`` changes to handle it today. - """ - - def __init__( - self, - *, - backend_factory: BackendFactory, - backend_name: str, - backend_config: dict[str, Any], - agent_config_path: str, - scenario_db_path: str, - current_date_time: str, - turn_end_fallback_seconds: float | None = None, - ) -> None: - """Initialize the assistant role. - - Args: - backend_factory: Factory used to construct the backend. - backend_name: Key passed to the factory to select a backend. - backend_config: Provider-specific configuration for the backend. - agent_config_path: Path to the agent YAML (role, instructions, - tool schemas) -- mirrors ``AbstractAssistantServer.agent`` / - ``agent_config_path``. - scenario_db_path: Path to the per-record scenario database JSON - consumed by tool execution -- mirrors - ``AbstractAssistantServer.scenario_db_path``. - current_date_time: Current date/time string threaded into both - prompt construction and tool execution (mirrors existing - ``current_date_time`` plumbing throughout the assistant - stack). - turn_end_fallback_seconds: How long after the assistant stops - speaking to wait for a user turn before firing a turn-end - fallback nudge, or ``None`` to disable the fallback entirely - (preserving the old behavior of waiting for the provider's - inactivity timeout). Mirrors the shipped - ``EVA_TURN_END_FALLBACK_TIME`` knob. This is an - ``AssistantRole``-level tuning value, not a - ``BackendCapabilities`` flag (capabilities describe what a - backend *can* do, statically). Wiring it into the constructed - ``self.backend``'s own config (via ``backend_config`` / - ``Backend.open(config=...)``) is left to the concrete - subclass's constructor, same as elsewhere in this contract -- - a ``Role`` does not otherwise reach into backend config after - construction. A backend with no notion of idle timing (e.g. a - thin end-to-end backend that relies on its own provider - backstop) may simply ignore this value. - """ - super().__init__(backend_factory=backend_factory, backend_name=backend_name, backend_config=backend_config) - self.agent_config_path = agent_config_path - self.scenario_db_path = scenario_db_path - self.current_date_time = current_date_time - self.turn_end_fallback_seconds = turn_end_fallback_seconds - - @abstractmethod - def get_final_scenario_db(self) -> dict[str, Any]: - """Return the (possibly mutated) scenario database state, for metrics. - - Mirrors ``AbstractAssistantServer.get_final_scenario_db()``. - """ - ... diff --git a/src/eva/role/base.py b/src/eva/role/base.py deleted file mode 100644 index a75b4fa1..00000000 --- a/src/eva/role/base.py +++ /dev/null @@ -1,159 +0,0 @@ -"""Abstract ``Role`` base contract: prompt/tools/goal ownership over a ``Backend``. - -DESIGN ONLY (Step 1 of the refactor, see docs/refactor-step1.md). Every -method body here is a stub -- this module defines shapes, not behavior, and -is not imported by any existing code path. - -This module holds only the shared ``Role`` base. The two concrete roles live -in sibling modules -- ``AssistantRole`` in ``eva.role.assistant`` and -``UserRole`` in ``eva.role.user`` -- since each carries a meaningfully -different data payload and plug-in point (see those modules' docstrings) and -is expected to grow its own implementation in later phases; keeping them in -separate files keeps each phase's diff scoped to one role. - -Design choice -- one ``Role`` base with ``AssistantRole``/``UserRole`` -subclasses, rather than two unrelated ABCs: - Both roles share an identical *control loop* shape: construct a backend - via ``BackendFactory``, ``build_prompt()`` before opening it, drive - ``backend.receive()`` and dispatch tool-call requests to - ``handle_tool_call_request()``, and record recorded audio/transcript for - output. What differs between them is only the *data* they carry (agent - config + tool catalog for the assistant; goal + persona + starting - utterance for the user) and how they decide the conversation is over. - That's a difference in constructor args and a couple of abstract methods, - not in control flow -- so one shared base with two thin subclasses avoids - duplicating the event loop, while still keeping tool-ownership and - prompt-building role-specific via abstract methods. If the two roles' - control loops diverge significantly in a later phase, splitting them - apart is a mechanical extraction of ``Role`` into two ABCs -- nothing - here should make that harder. -""" - -from __future__ import annotations - -from abc import ABC, abstractmethod -from pathlib import Path -from typing import Any - -from eva.backend.base import Backend, ToolCallRequest, ToolCallResult -from eva.backend.factory import BackendFactory - - -class Role(ABC): - """Owns prompt, tools/goal, and a runtime-created ``Backend``. - - A ``Role`` is the thing that used to be split across - ``AbstractAssistantServer`` (assistant side) and ``AbstractUserSimulator`` - (user side): everything that is *not* pure provider API exchange lives - here instead of in ``Backend``. In particular: - - - Tool execution stays role-side (per docs/refactor-step1.md): a ``Role`` - is responsible for turning a ``ToolCallRequest`` surfaced by its - backend into a ``ToolCallResult``, using whatever execution engine is - appropriate for that role (``ToolExecutor`` for ``AssistantRole``; a - trivial/no-op handler for ``UserRole``, which today only exposes a - synthetic ``end_call`` tool). ``Backend`` implementations never - execute tools themselves. - - Prompt construction stays role-side: ``build_prompt()`` replaces both - ``AbstractAssistantServer._build_system_prompt()`` / - ``AgenticSystem``'s prompt building and - ``AbstractUserSimulator._build_prompt()``. - - Audio recording / output persistence is a role-side concern shared by - both subclasses (see ``docs/refactor-step1.md`` point 5, "consolidate - audio recording / output-saving into one shared helper") -- this base - class declares the seam (``record_audio`` / ``save_outputs``) but does - not implement the shared helper itself; that helper is later work. - """ - - def __init__(self, *, backend_factory: BackendFactory, backend_name: str, backend_config: dict[str, Any]) -> None: - """Construct the role's backend (but do not open its session yet). - - Args: - backend_factory: Factory used to construct ``self.backend``. - backend_name: Provider name passed through to - ``BackendFactory.create``. - backend_config: Provider-specific config passed through to - ``BackendFactory.create`` (not to be confused with the - ``config`` argument of ``Backend.open``, which is also - provider-specific but may be augmented by the role at - open-time, e.g. with a resolved sample rate). - """ - self.backend: Backend = backend_factory.create(backend_name, backend_config) - - @abstractmethod - def build_prompt(self) -> str: - """Build the full system prompt / instructions for this role. - - For ``AssistantRole`` this replaces - ``AbstractAssistantServer._build_system_prompt()`` and - ``AgenticSystem``'s inline prompt construction. For ``UserRole`` this - replaces ``AbstractUserSimulator._build_prompt()``. Called before - ``Backend.open()`` so the result can be passed as its - ``system_prompt`` argument. - """ - ... - - @abstractmethod - async def handle_tool_call_request(self, request: ToolCallRequest) -> ToolCallResult: - """Execute a tool call surfaced by this role's backend and return the result. - - This is the single place tool execution happens for this role -- - ``Backend`` implementations must never execute tools directly (see - class docstring). Implementations should log the call/result (e.g. - to an audit log) as part of executing it. - """ - ... - - @abstractmethod - async def run(self) -> str: - """Drive the conversation for this role until it reaches a terminal state. - - Expected shape (left to subclasses to implement, not prescribed in - detail here since the exact loop depends on the backend's - capabilities -- see ``BackendCapabilities``): - 1. ``await self.backend.open(system_prompt=self.build_prompt(), ...)`` - 2. Iterate ``self.backend.receive()``, dispatching - ``TOOL_CALL_REQUEST`` events to ``handle_tool_call_request`` and - feeding the ``ToolCallResult`` back via - ``self.backend.send(tool_result=...)``. - 3. Record audio/transcript events as they arrive (see - ``record_audio``). - 4. On a terminal event (hangup, timeout, transfer, error), call - ``await self.backend.close()`` and return an end-reason string. - - Returns: - A short end-reason string (e.g. ``"goodbye"``, ``"transfer"``, - ``"timeout"``, ``"error"``) -- mirrors the return contract of - today's ``AbstractUserSimulator.run_conversation()``. - """ - ... - - @abstractmethod - def record_audio(self, source: str, audio_data: bytes) -> None: - """Accumulate a chunk of audio for later persistence. - - Args: - source: Role-defined stream label (e.g. ``"user"``, - ``"assistant"``, or a cleaned/pre-perturbation variant). - Mirrors ``AbstractUserSimulator._record_audio`` / - ``AbstractAssistantServer``'s audio-buffer fields; the exact - set of valid labels is left to subclasses/shared helper, not - fixed by this contract. - audio_data: Raw PCM16 bytes at this role's recording sample rate. - """ - ... - - @abstractmethod - async def save_outputs(self, output_dir: Path) -> None: - """Persist this role's output artifacts to ``output_dir``. - - For ``AssistantRole`` this covers ``audit_log.json``, - ``transcript.jsonl``, scenario DB snapshots (mirrors - ``AbstractAssistantServer.save_outputs``). For ``UserRole`` this - covers ``user_simulator_events.jsonl`` (mirrors the event logger in - ``AbstractUserSimulator``). Audio WAV files are expected to be - written by the shared audio-recording helper referenced in - ``record_audio``, not necessarily by this method -- exact division of - labor is left to the later implementation phase. - """ - ... diff --git a/src/eva/role/user.py b/src/eva/role/user.py deleted file mode 100644 index 415d8b7a..00000000 --- a/src/eva/role/user.py +++ /dev/null @@ -1,83 +0,0 @@ -"""``UserRole`` contract: the simulated-caller role. - -DESIGN ONLY (Step 1 of the refactor, see docs/refactor-step1.md). Method -bodies are stubs; nothing here is wired into the existing code path yet. - -Plug-in point (where this will eventually replace existing code): - Today the user side is a concrete ``AbstractUserSimulator`` subclass - selected by ``eva.user_simulator.factory.create_user_simulator(config, ...)`` - and constructed inside ``ConversationWorker._start_user_simulator()`` - (worker.py), which passes it ``server_url=f"ws://localhost:{port}/ws"`` - to reach the assistant server. The conversation is driven by - ``ConversationWorker._run_conversation()`` calling - ``user_simulator.run_conversation()``, whose returned end-reason string - becomes the conversation result. - - In a later phase, ``_start_user_simulator()`` becomes the construction - site for a ``UserRole`` (simulator config -> ``backend_name`` + - ``backend_config`` for the ``BackendFactory``), and the worker drives - ``role.run()`` (returning the same end-reason string via - ``get_end_reason()``) instead of ``run_conversation()``. Note the - ``server_url`` handoff is a *transport* detail that today's user side owns - directly; per docs/refactor-step1.md the ``Backend`` contract is kept - direction-agnostic precisely so this WS-connect concern can move into a - ``Backend`` implementation (or, later, a mediator) without the role - caring. This module is deliberately separate from ``eva.role.assistant`` - so the user-side migration can land as its own scoped diff. -""" - -from __future__ import annotations - -from abc import abstractmethod -from typing import Any - -from eva.backend.factory import BackendFactory -from eva.role.base import Role - - -class UserRole(Role): - """Role that simulates the human caller (today's "user simulator"). - - Carries goal/persona instead of agent config/tools -- its tool surface, - if any, is limited to caller-side affordances like ``end_call`` (see - ``END_CALL_DESCRIPTION`` in today's ``eva.user_simulator.base``), not a - business tool catalog. - """ - - def __init__( - self, - *, - backend_factory: BackendFactory, - backend_name: str, - backend_config: dict[str, Any], - goal: dict[str, Any], - persona_config: dict[str, Any], - current_date_time: str, - ) -> None: - """Initialize the user role. - - Args: - backend_factory: Factory used to construct the backend. - backend_name: Key passed to the factory to select a backend. - backend_config: Provider-specific configuration for the backend. - goal: User goal / decision-tree data -- mirrors - ``AbstractUserSimulator.goal``. - persona_config: Persona/voice/behavior configuration -- mirrors - ``AbstractUserSimulator.persona_config``. - current_date_time: Threaded into prompt construction, mirroring - existing plumbing. - """ - super().__init__(backend_factory=backend_factory, backend_name=backend_name, backend_config=backend_config) - self.goal = goal - self.persona_config = persona_config - self.current_date_time = current_date_time - - @abstractmethod - def get_end_reason(self) -> str: - """Return the terminal end-reason for this conversation. - - Mirrors the return value of today's - ``AbstractUserSimulator.run_conversation()`` (``"goodbye"``, - ``"transfer"``, ``"timeout"``, ``"error"``, ...). - """ - ... diff --git a/src/eva/user_simulator/cascade/__init__.py b/src/eva/user_simulator/cascade/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/src/eva/user_simulator/cascade/adapter/__init__.py b/src/eva/user_simulator/cascade/adapter/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/src/eva/user_simulator/cascade/adapter/base.py b/src/eva/user_simulator/cascade/adapter/base.py new file mode 100644 index 00000000..c81914c6 --- /dev/null +++ b/src/eva/user_simulator/cascade/adapter/base.py @@ -0,0 +1,32 @@ +"""Adapter contract: the only component doing real I/O against the assistant.""" + +from __future__ import annotations + +from abc import ABC, abstractmethod + +from eva.user_simulator.cascade.tick_result import TickResult + + +class Adapter(ABC): + """Exchanges exactly one tick of audio with the assistant per call. + + The scheduler and simulator never learn which implementation is behind this + interface, which is what lets a tick-driven transport replace the real-time + one without touching turn-taking logic. + """ + + @abstractmethod + async def start(self) -> None: + """Establish the connection. Must return once ready to exchange audio.""" + + @abstractmethod + async def run_tick(self, tick_number: int, outgoing_audio: bytes | None, *, barge_in: bool = False) -> TickResult: + """Send this tick's caller audio (None means silence) and collect what arrived. + + ``barge_in`` signals that this tick begins an interruption. Adapters that + cannot truncate provider-side audio ignore it. + """ + + @abstractmethod + async def stop(self) -> None: + """Tear the connection down. Must be safe to call twice.""" diff --git a/src/eva/user_simulator/cascade/adapter/realtime_ws.py b/src/eva/user_simulator/cascade/adapter/realtime_ws.py new file mode 100644 index 00000000..4dd3e817 --- /dev/null +++ b/src/eva/user_simulator/cascade/adapter/realtime_ws.py @@ -0,0 +1,228 @@ +"""Real-time Twilio WebSocket adapter for unmodified assistant servers.""" + +from __future__ import annotations + +import asyncio +import base64 +import contextlib +import json +import time + +try: + import audioop +except ImportError: # pragma: no cover - Python 3.13+ + import audioop_lts as audioop + +from eva.user_simulator.cascade.adapter.base import Adapter +from eva.user_simulator.cascade.constants import BYTES_PER_TICK, CALLER_SAMPLE_RATE, SILENCE_BYTE, TICK_DURATION_MS +from eva.user_simulator.cascade.tick_result import TickResult, split_tick_audio +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + +WIRE_SAMPLE_RATE = 8000 +WIRE_FRAME_MS = 20 +FRAMES_PER_TICK = TICK_DURATION_MS // WIRE_FRAME_MS +_PCM_WIDTH = 2 + + +class RealtimeWSAdapter(Adapter): + """Exchanges tick-sized audio with an assistant server over the Twilio WS protocol. + + Outbound audio is paced at the real 20ms cadence the assistant expects + (docs/assistant_server_contract.md section 3), and every tick sends a full + tick's worth of frames — real audio or synthesized silence — so the assistant's + STT/VAD always sees an unbroken stream, the way a real phone line would; gaps + with no frames at all are what caused turn detection to misfire. A background + task continuously drains inbound frames into an adapter-owned buffer; `run_tick` + releases exactly one tick's worth per call, so a provider that generates faster + than real time cannot run ahead of the simulation clock. `run_tick` also enforces + a minimum tick duration as a safety net for ticks that send quickly. + + Each direction resamples a continuous stream, so each carries its own + `audioop.ratecv` filter state across calls; the stateless helpers in + `audio_utils` cannot express that and are not reused here for that reason. + """ + + def __init__( + self, + *, + websocket, + conversation_id: str, + bytes_per_tick: int = BYTES_PER_TICK, + perturbator=None, + ) -> None: + self._ws = websocket + self._conversation_id = conversation_id + self._bytes_per_tick = bytes_per_tick + self._perturbator = perturbator + self._inbound = bytearray() + self._receive_task: asyncio.Task | None = None + self._inbound_resample_state = None + self._outbound_resample_state = None + self._caller_speaking = False + self._error: BaseException | None = None + + async def start(self) -> None: + """Send the connect/start handshake and begin buffering inbound audio.""" + for event in ("connected", "start"): + await self._ws.send(json.dumps({"event": event, "conversation_id": self._conversation_id})) + self._receive_task = asyncio.create_task(self._receive_loop()) + + async def run_tick(self, tick_number: int, outgoing_audio: bytes | None, *, barge_in: bool = False) -> TickResult: + """Send one tick of caller audio at wire cadence and collect one tick of assistant audio. + + ``barge_in`` is accepted and ignored: this assistant paces its own output, so + it has generated nothing past what the caller already heard to discard. + """ + tick_start = asyncio.get_event_loop().time() + if self._error is not None: + raise RuntimeError("RealtimeWSAdapter receive loop failed") from self._error + + is_speaking = bool(outgoing_audio) + if is_speaking and not self._caller_speaking: + await self._send_speech_event("user_speech_start") + elif not is_speaking and self._caller_speaking: + await self._send_speech_event("user_speech_stop") + self._caller_speaking = is_speaking + + outgoing = self._apply_perturbation(outgoing_audio) + await self._send_tick_audio(outgoing or SILENCE_BYTE * self._bytes_per_tick) + + raw = bytes(self._inbound[: self._bytes_per_tick]) + del self._inbound[: len(raw)] + chunk, _ = split_tick_audio(raw, self._bytes_per_tick) + + if self._error is not None: + raise RuntimeError("RealtimeWSAdapter receive loop failed") from self._error + + result = TickResult( + tick_number=tick_number, + assistant_audio=chunk, + assistant_audio_raw_bytes=len(raw), + wall_clock_ms=int(time.time() * 1000), + ) + + remaining = TICK_DURATION_MS / 1000 - (asyncio.get_event_loop().time() - tick_start) + if remaining > 0: + await asyncio.sleep(remaining) + + return result + + def _apply_perturbation(self, outgoing_audio: bytes | None) -> bytes | None: + """Mix ambient noise into this tick's outgoing audio. + + Real-time path: the mic is always open, so noise is emitted even when the + caller is silent — it replaces the silence this adapter already sends every + tick rather than adding frames. Never call this for a tick that emits nothing + on the tick-driven path (Plan 3): audio sent during a stall would advance the + assistant's VAD and break the freeze. + """ + if self._perturbator is None: + return outgoing_audio + if outgoing_audio: + return self._perturbator.apply(outgoing_audio) + if getattr(self._perturbator, "has_ambient_noise", False): + return self._perturbator.get_ambient_chunk(self._bytes_per_tick) + return outgoing_audio + + async def stop(self) -> None: + """Send stop, cancel the receive loop, and close the socket. Safe to call twice.""" + if self._receive_task is not None: + task = self._receive_task + self._receive_task = None + task.cancel() + try: + await task + except asyncio.CancelledError: + if not task.cancelled(): + raise + except Exception: + pass + with contextlib.suppress(Exception): + await self._ws.send(json.dumps({"event": "stop", "conversation_id": self._conversation_id})) + with contextlib.suppress(Exception): + await self._ws.close() + + async def _send_speech_event(self, event: str) -> None: + """Emit a user_speech_start/stop event matching the existing bridge's payload shape.""" + await self._ws.send( + json.dumps( + { + "event": event, + "conversation_id": self._conversation_id, + "timestamp_ms": str(int(round(time.time() * 1000))), + } + ) + ) + + async def _send_tick_audio(self, pcm: bytes) -> None: + """Split one tick of PCM16 into 20ms mulaw frames and send them at real-time pace.""" + mulaw = self._pcm16k_to_mulaw8k(pcm) + if not mulaw: + return + frame_size = len(mulaw) // FRAMES_PER_TICK + frames = [mulaw[index : index + frame_size] for index in range(0, len(mulaw), frame_size)] + interval = WIRE_FRAME_MS / 1000 + start_time = asyncio.get_event_loop().time() + for frame_index, frame in enumerate(frames): + payload = base64.b64encode(frame).decode() + await self._ws.send( + json.dumps( + { + "event": "media", + "conversation_id": self._conversation_id, + "media": {"payload": payload}, + } + ) + ) + if frame_index == len(frames) - 1: + break + deadline = start_time + (frame_index + 1) * interval + sleep_for = deadline - asyncio.get_event_loop().time() + if sleep_for > 0: + await asyncio.sleep(sleep_for) + + async def _receive_loop(self) -> None: + """Continuously buffer inbound assistant audio as PCM16 at the caller sample rate.""" + while True: + try: + raw = await self._ws.recv() + except asyncio.CancelledError: + raise + except Exception as exc: + logger.exception("RealtimeWSAdapter receive loop failed") + self._error = exc + return + self._ingest(raw) + + def _ingest(self, raw: str) -> None: + """Decode one inbound frame, keeping only media payloads.""" + try: + message = json.loads(raw) + except json.JSONDecodeError: + return + if message.get("event") != "media": + return + payload = message.get("media", {}).get("payload", "") + if payload: + self._inbound.extend(self._mulaw8k_to_pcm16k(base64.b64decode(payload))) + self._on_inbound_audio() + + def _on_inbound_audio(self) -> None: + """Hook fired after inbound audio lands in the buffer. No-op on this path.""" + + def _mulaw8k_to_pcm16k(self, mulaw: bytes) -> bytes: + """Convert 8kHz mulaw from the wire to PCM16 at the caller sample rate.""" + pcm_8k = audioop.ulaw2lin(mulaw, _PCM_WIDTH) + pcm_16k, self._inbound_resample_state = audioop.ratecv( + pcm_8k, _PCM_WIDTH, 1, WIRE_SAMPLE_RATE, CALLER_SAMPLE_RATE, self._inbound_resample_state + ) + return pcm_16k + + def _pcm16k_to_mulaw8k(self, pcm: bytes) -> bytes: + """Convert caller PCM16 to the 8kHz mulaw the assistant expects.""" + pcm_8k, self._outbound_resample_state = audioop.ratecv( + pcm, _PCM_WIDTH, 1, CALLER_SAMPLE_RATE, WIRE_SAMPLE_RATE, self._outbound_resample_state + ) + return audioop.lin2ulaw(pcm_8k, _PCM_WIDTH) diff --git a/src/eva/user_simulator/cascade/adapter/tick_driven.py b/src/eva/user_simulator/cascade/adapter/tick_driven.py new file mode 100644 index 00000000..7c0944de --- /dev/null +++ b/src/eva/user_simulator/cascade/adapter/tick_driven.py @@ -0,0 +1,197 @@ +"""Tick-driven adapter: the caller owns the simulation clock. + +Talks to the same assistant server over the same Twilio WebSocket as the +real-time adapter, and sends a full tick of audio — real or silence — on every +tick just as it does. The difference is timing: outbound frames carry no pacing +sleeps, because the server is not pacing its own output either +(``paced_output=False``), and there is no minimum tick duration. + +Because the provider's VAD advances on audio received rather than on wall time, +the conversation advances only when the caller ticks it. While the caller is +generating a turn no tick runs, so no audio flows and the assistant is frozen — +that is what makes caller compute time invisible. The freeze comes from *not +ticking*, not from emitting nothing within a tick: a tick that sent nothing would +starve the VAD of the trailing silence that ends the caller's turn, and the +assistant would never reply at all. + +Inbound audio is released strictly one tick at a time however much arrives at +once, which the real-time adapter already does; the difference is that here the +release cadence *is* the simulation clock rather than an approximation of it. + +Implemented as a subclass of :class:`RealtimeWSAdapter` rather than a peer: the +handshake, the receive loop, the speech-event frames and — critically — the +mulaw/PCM resamplers all carry per-instance filter state that must not be +duplicated. Only the timing is overridden. +""" + +from __future__ import annotations + +import asyncio +import base64 +import json +import time + +from eva.user_simulator.cascade.adapter.realtime_ws import FRAMES_PER_TICK, RealtimeWSAdapter +from eva.user_simulator.cascade.constants import BYTES_PER_TICK, SILENCE_BYTE, TICK_DURATION_MS +from eva.user_simulator.cascade.tick_result import TickResult, played_audio_ms, split_tick_audio +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + +MAX_INACTIVE_SECONDS = 40.0 +"""Fail loudly if the provider goes quiet this long (tau: DEFAULT_AUDIO_NATIVE_MAX_INACTIVE_SECONDS).""" + +QUIET_TICK_GRACE_S = TICK_DURATION_MS / 1000 +"""How long a tick waits for a full tick of assistant audio before calling it silence. + +Not pacing: it never *delays* audio that has already arrived, so a provider +generating faster than real time is still drained as fast as it produces, and +caller compute still costs the conversation nothing. It is only the bound on how +long "the assistant has said nothing yet" takes to establish. Without it the loop +spins through its whole tick budget in milliseconds and every conversation ends at +tick zero with nothing said.""" + + +class TickDrivenAdapter(RealtimeWSAdapter): + """Exchanges one tick of audio with an unpaced assistant server.""" + + def __init__( + self, + *, + websocket, + conversation_id: str, + bytes_per_tick: int = BYTES_PER_TICK, + perturbator=None, + ) -> None: + super().__init__( + websocket=websocket, + conversation_id=conversation_id, + bytes_per_tick=bytes_per_tick, + perturbator=perturbator, + ) + self._ticks_released = 0 + self._last_inbound_monotonic = time.monotonic() + self._audio_arrived = asyncio.Event() + + @property + def played_ms(self) -> int: + """Assistant audio released into the conversation so far, in simulated ms.""" + return played_audio_ms(ticks_released=self._ticks_released) + + async def run_tick(self, tick_number: int, outgoing_audio: bytes | None, *, barge_in: bool = False) -> TickResult: + """Send this tick's caller audio unpaced and release exactly one tick of assistant audio. + + When ``barge_in`` is set, first tell the assistant to discard the audio it + generated past the position the caller has actually heard. + """ + if self._error is not None: + raise RuntimeError("TickDrivenAdapter receive loop failed") from self._error + + interruption_start: int | None = None + if barge_in: + interruption_start = self.played_ms + await self._ws.send( + json.dumps( + { + "event": "truncate", + "conversation_id": self._conversation_id, + "audio_end_ms": interruption_start, + } + ) + ) + # Everything already buffered is audio the caller never heard. + self._inbound.clear() + + is_speaking = bool(outgoing_audio) + if is_speaking and not self._caller_speaking: + await self._send_speech_event("user_speech_start") + elif not is_speaking and self._caller_speaking: + await self._send_speech_event("user_speech_stop") + self._caller_speaking = is_speaking + + # Every tick puts a full tick on the wire, silence included, exactly as the + # real-time adapter does. The provider's VAD advances on audio *received*, so + # a tick that emits nothing never ends the caller's turn and the assistant + # never replies. The freeze this plan is built on comes from the caller not + # calling `run_tick` while it thinks — not from emitting nothing inside one. + outgoing = self._apply_perturbation(outgoing_audio) + await self._send_unpaced(outgoing or SILENCE_BYTE * self._bytes_per_tick) + + await self._await_tick_of_audio() + raw = bytes(self._inbound[: self._bytes_per_tick]) + del self._inbound[: len(raw)] + chunk, _ = split_tick_audio(raw, self._bytes_per_tick) + if raw: + self._ticks_released += 1 + + stalled = self._provider_has_stalled(bool(raw)) + + if self._error is not None: + raise RuntimeError("TickDrivenAdapter receive loop failed") from self._error + + return TickResult( + tick_number=tick_number, + assistant_audio=chunk, + assistant_audio_raw_bytes=len(raw), + wall_clock_ms=int(time.time() * 1000), + interruption_audio_start_ms=interruption_start, + provider_stalled=stalled, + ) + + def _on_inbound_audio(self) -> None: + """Wake a tick that is waiting on assistant audio.""" + self._audio_arrived.set() + + async def _await_tick_of_audio(self) -> None: + """Wait until a whole tick of assistant audio is buffered, or the grace expires. + + Returns as soon as the buffer is full, so nothing already generated is held + back. The grace only bounds the silent case. + """ + deadline = time.monotonic() + QUIET_TICK_GRACE_S + while len(self._inbound) < self._bytes_per_tick: + remaining = deadline - time.monotonic() + if remaining <= 0: + return + self._audio_arrived.clear() + if len(self._inbound) >= self._bytes_per_tick: + return + try: + await asyncio.wait_for(self._audio_arrived.wait(), timeout=remaining) + except TimeoutError: + return + + async def _send_unpaced(self, pcm: bytes) -> None: + """Split one tick of PCM16 into wire frames and send them with no sleeps.""" + mulaw = self._pcm16k_to_mulaw8k(pcm) + if not mulaw: + return + frame_size = len(mulaw) // FRAMES_PER_TICK or len(mulaw) + for index in range(0, len(mulaw), frame_size): + payload = base64.b64encode(mulaw[index : index + frame_size]).decode() + await self._ws.send( + json.dumps( + { + "event": "media", + "conversation_id": self._conversation_id, + "media": {"payload": payload}, + } + ) + ) + + def _provider_has_stalled(self, received: bool) -> bool: + """Whether the provider has produced nothing for too long. + + Wall clock is the right measure here and only here: this is a liveness + check on a real network peer, not a measurement of conversation time. + + Reported, not raised. Raising aborted the tick loop from underneath the + simulator, so the record ended on the generic "error" reason with no way to + tell a stalled provider from a bug in the caller, and the terminal state the + runner reads was never written. The caller ends the conversation instead. + """ + now = time.monotonic() + if received: + self._last_inbound_monotonic = now + return False + return now - self._last_inbound_monotonic > MAX_INACTIVE_SECONDS diff --git a/src/eva/user_simulator/cascade/constants.py b/src/eva/user_simulator/cascade/constants.py new file mode 100644 index 00000000..94e5c930 --- /dev/null +++ b/src/eva/user_simulator/cascade/constants.py @@ -0,0 +1,44 @@ +"""Timing constants for the tick-based cascade caller. + +Values mirror tau-voice (tau2-bench/src/tau2/config.py:104-113). These are +module constants rather than config fields on purpose: there is no run-level +reason to vary them, and exposing them would let runs drift apart in ways that +make their metrics incomparable. +""" + +TICK_DURATION_MS = 200 +"""One simulation tick: 200ms of PCM16 audio at CALLER_SAMPLE_RATE, converted by +the adapter into ten 20ms mulaw frames when written to the wire.""" + +WAIT_TO_RESPOND_OTHER_MS = 1000 +"""Silence required from the assistant before the caller starts a turn.""" + +WAIT_TO_RESPOND_SELF_MS = 5000 +"""Silence required from the caller itself before it starts another turn.""" + +TRANSCRIPT_WAIT_MS = 1500 +"""How long the caller waits for the assistant's transcript to finalize before falling back +to the in-flight partial, sized above the slowest measured finalization (ink-2, 1.2s).""" + +INACTIVITY_TIMEOUT_MS = 120000 +"""Assistant silence that ends the conversation, matching ElevenLabsUserSimulator's 12 +keep-alives so both providers record the same inactivity_timeout terminal state.""" + +CALLER_SAMPLE_RATE = 16000 +"""PCM16 sample rate for the caller's own audio track.""" + +_BYTES_PER_SAMPLE = 2 + +BYTES_PER_TICK = CALLER_SAMPLE_RATE * TICK_DURATION_MS // 1000 * _BYTES_PER_SAMPLE +"""PCM16 bytes carried per tick at CALLER_SAMPLE_RATE.""" + +SILENCE_BYTE = b"\x00" +"""PCM16 silence, used to pad partial ticks.""" + +LISTENER_CHECK_INTERVAL_MS = 2000 +"""How often the interrupt and backchannel checks run while the assistant speaks.""" + + +def ms_to_ticks(milliseconds: int) -> int: + """Convert milliseconds to whole ticks, flooring.""" + return milliseconds // TICK_DURATION_MS diff --git a/src/eva/user_simulator/cascade/decision_log.py b/src/eva/user_simulator/cascade/decision_log.py new file mode 100644 index 00000000..f9cc29a7 --- /dev/null +++ b/src/eva/user_simulator/cascade/decision_log.py @@ -0,0 +1,58 @@ +"""Per-tick diagnostic trace for the cascade caller's out-of-turn decisions.""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, TextIO + +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + + +class DecisionLog: + """Records every listener check, including the ones that declined or never ran. + + `user_simulator_events.jsonl` only carries actions the caller took, so a check that + ran and said NO is indistinguishable there from a check that never fired. This trace + separates the two, which is the difference between "the model does not want to + interrupt" and "the interrupt path is unreachable". + + Rows are written and flushed as they happen rather than buffered to the end: a + conversation that dies mid-run is exactly the one worth having a trace for, and the + event log's save-at-exit is why failed attempts currently leave no diagnostics at all. + """ + + def __init__(self, output_path: Path) -> None: + self.output_path = output_path + self._handle: TextIO | None = None + self._counts: dict[str, int] = {} + + def log(self, kind: str, **fields: Any) -> None: + """Append one trace row of the given kind and flush it to disk.""" + self._counts[kind] = self._counts.get(kind, 0) + 1 + try: + handle = self._open() + handle.write(json.dumps({"kind": kind, **fields}, ensure_ascii=False) + "\n") + handle.flush() + except Exception as exc: + logger.warning(f"Could not write caller decision trace: {exc}") + + def _open(self) -> TextIO: + """Open the trace on first use, so a run that traces nothing leaves no file.""" + if self._handle is None: + self.output_path.parent.mkdir(parents=True, exist_ok=True) + self._handle = open(self.output_path, "w") + return self._handle + + def save(self) -> None: + """Close the trace. Safe to call more than once.""" + if self._handle is not None: + self._handle.close() + self._handle = None + logger.info(f"Caller decision trace written to {self.output_path}") + + def summary(self) -> dict[str, int]: + """Count rows by kind, for a one-line end-of-run report.""" + return dict(self._counts) diff --git a/src/eva/user_simulator/cascade/decisions.py b/src/eva/user_simulator/cascade/decisions.py new file mode 100644 index 00000000..6afedf3d --- /dev/null +++ b/src/eva/user_simulator/cascade/decisions.py @@ -0,0 +1,95 @@ +"""Listener-reaction checks: should the caller interrupt or backchannel right now.""" + +from __future__ import annotations + +import asyncio +import time +from dataclasses import dataclass, field +from typing import Protocol + +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + + +class DecisionLLM(Protocol): + """Minimal interface the checks need from a model client.""" + + async def decide(self, prompt: str) -> str: + """Return the model's raw reply to a YES/NO question.""" + ... + + +@dataclass(frozen=True) +class CheckTrace: + """What one YES/NO check actually did, for the diagnostic trace.""" + + ran: bool + raw: str = "" + latency_ms: int = 0 + error: str = "" + + +@dataclass(frozen=True) +class ListenerVerdict: + """Outcome of one check tick.""" + + should_interrupt: bool + should_backchannel: bool + interrupt_trace: CheckTrace = field(default_factory=lambda: CheckTrace(ran=False)) + backchannel_trace: CheckTrace = field(default_factory=lambda: CheckTrace(ran=False)) + + +def parse_yes_no(raw: str) -> bool: + """Read a bare YES/NO reply. Anything unrecognized counts as NO.""" + return raw.strip().upper() == "YES" + + +class ListenerDecisions: + """Runs the interrupt and backchannel checks concurrently against the partial transcript. + + Both fail closed: an exception yields "don't act", so a provider hiccup can + never inject a barge-in that the caller never actually decided to make. + """ + + def __init__( + self, llm: DecisionLLM, *, interrupt_prompt: str, backchannel_prompt: str, user_goal: str = "" + ) -> None: + self._llm = llm + self._interrupt_prompt = interrupt_prompt + self._backchannel_prompt = backchannel_prompt + self._user_goal = user_goal + + async def evaluate( + self, conversation_history: str, *, allow_interrupt: bool, allow_backchannel: bool + ) -> ListenerVerdict: + """Run whichever checks are enabled. Interrupt wins ties (tau: streaming.py:2549).""" + interrupt_trace, backchannel_trace = await asyncio.gather( + self._check(self._interrupt_prompt, conversation_history, enabled=allow_interrupt), + self._check(self._backchannel_prompt, conversation_history, enabled=allow_backchannel), + ) + interrupt = interrupt_trace.ran and parse_yes_no(interrupt_trace.raw) + backchannel = backchannel_trace.ran and parse_yes_no(backchannel_trace.raw) + return ListenerVerdict( + should_interrupt=interrupt, + should_backchannel=backchannel and not interrupt, + interrupt_trace=interrupt_trace, + backchannel_trace=backchannel_trace, + ) + + async def _check(self, template: str, conversation_history: str, *, enabled: bool) -> CheckTrace: + """Ask the model one YES/NO question, reporting what happened rather than just the answer. + + Both templates are filled with the same arguments; `str.format` ignores the ones a + given prompt does not use, so the backchannel prompt needs no goal slot. + """ + if not enabled: + return CheckTrace(ran=False) + started = time.monotonic() + try: + filled = template.format(conversation_history=conversation_history, user_goal=self._user_goal) + reply = await self._llm.decide(filled) + except Exception as exc: + logger.warning(f"Listener check failed, defaulting to no action: {exc}") + return CheckTrace(ran=True, latency_ms=int((time.monotonic() - started) * 1000), error=str(exc)) + return CheckTrace(ran=True, raw=reply, latency_ms=int((time.monotonic() - started) * 1000)) diff --git a/src/eva/user_simulator/cascade/phrase_cache.py b/src/eva/user_simulator/cascade/phrase_cache.py new file mode 100644 index 00000000..69d7e966 --- /dev/null +++ b/src/eva/user_simulator/cascade/phrase_cache.py @@ -0,0 +1,84 @@ +"""Pre-rendered audio for the caller's fixed phrase vocabularies.""" + +from __future__ import annotations + +import asyncio +import random +from typing import ClassVar, Protocol + +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + + +class SpeechSynthesizer(Protocol): + """Minimal interface the cache needs from a TTS client.""" + + async def synthesize(self, text: str, *, voice_id: str) -> bytes: + """Render text to PCM16 audio.""" + ... + + +class PhraseCache: + """Renders a fixed vocabulary once so it can be voiced at zero latency. + + Backchannels and barge-in openers are short, fixed, and voice-stable, which + makes them cacheable — and caching is the only way a 300ms "mm-hmm" reliably + lands on the tick the decision chose. + + The audio is held **per process, keyed by (voice_id, phrase)**, not per + simulator. The vocabulary depends only on the voice, so a run of N records + against two voices needs two renders of each phrase rather than 2N: every + conversation after the first finds the audio already there and starts without + waiting on TTS at all. Only the RNG is per-instance, so each conversation + still chooses its phrases independently and reproducibly. + """ + + _audio: ClassVar[dict[tuple[str, str], bytes]] = {} + _render_lock: ClassVar[asyncio.Lock | None] = None + + def __init__(self, tts: SpeechSynthesizer, *, voice_id: str, seed: int = 0) -> None: + self._tts = tts + self._voice_id = voice_id + self._rng = random.Random(seed) + + @classmethod + def _lock(cls) -> asyncio.Lock: + """Lazily create the shared render lock, on whichever loop is running.""" + if cls._render_lock is None: + cls._render_lock = asyncio.Lock() + return cls._render_lock + + @classmethod + def clear(cls) -> None: + """Drop all cached audio. For tests, and for a voice set changing mid-process.""" + cls._audio.clear() + + async def prerender(self, phrases: list[str]) -> None: + """Synthesize any phrase this voice has not rendered yet. + + Serialized across conversations: concurrent records share one voice, so + without the lock each would discover the same empty cache and render the + same phrases. The lock is held only for genuine misses. + """ + async with self._lock(): + missing = [p for p in phrases if (self._voice_id, p) not in self._audio] + if not missing: + logger.debug(f"All {len(phrases)} caller phrases already rendered for {self._voice_id}") + return + rendered = await asyncio.gather(*(self._tts.synthesize(p, voice_id=self._voice_id) for p in missing)) + self._audio.update( + {(self._voice_id, phrase): audio for phrase, audio in zip(missing, rendered, strict=True)} + ) + logger.info(f"Pre-rendered {len(missing)} caller phrases for voice {self._voice_id}") + + def get(self, phrase: str) -> bytes: + """Return cached audio for a phrase in this cache's voice.""" + key = (self._voice_id, phrase) + if key not in self._audio: + raise KeyError(f"Phrase not pre-rendered for voice {self._voice_id}: {phrase!r}") + return self._audio[key] + + def choose(self, phrases: list[str]) -> str: + """Pick a phrase using the cache's seeded RNG, so runs stay reproducible.""" + return self._rng.choice(phrases) diff --git a/src/eva/user_simulator/cascade/phrases.py b/src/eva/user_simulator/cascade/phrases.py new file mode 100644 index 00000000..6031d92d --- /dev/null +++ b/src/eva/user_simulator/cascade/phrases.py @@ -0,0 +1,102 @@ +"""Per-language phrase vocabularies for the caller's out-of-turn behavior. + +Data rather than constants because these are *words*, and words are language +specific. Timing thresholds stay in `constants.py` — varying those across runs +makes metrics incomparable, whereas speaking English continuers into a French +conversation is simply wrong. + +Generated per language by `scripts/add_culture_data.py`, alongside the initial +message it already translates, so a run only ever reads this file. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import yaml + +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + +PHRASES_PATH = Path(__file__).resolve().parents[4] / "configs" / "caller_phrases.yaml" + +FALLBACK_LANGUAGE = "en" + +_cache: dict[str, CallerPhrases] = {} + + +@dataclass(frozen=True) +class CallerPhrases: + """The fixed things the caller can say without taking a turn.""" + + backchannels: list[str] + """Continuers: "I'm listening, keep going".""" + + barge_in_openers: list[str] + """Opening fragment of an interruption, voiced ahead of the content.""" + + @property + def vocabulary(self) -> list[str]: + """Every phrase that needs pre-rendering, in a stable order.""" + return [*self.backchannels, *self.barge_in_openers] + + +def _read_file() -> dict[str, Any]: + """Load the phrase file, or return empty when it is missing or unreadable.""" + try: + data = yaml.safe_load(PHRASES_PATH.read_text(encoding="utf-8")) + except OSError as exc: + logger.warning(f"Caller phrase file unreadable ({exc}); out-of-turn behavior has no vocabulary") + return {} + return data or {} + + +def candidate_languages(language: str) -> list[str]: + """Language tags to try, most specific first: 'fr-CA' -> 'fr-CA', 'fr', 'en'. + + A regional variant nearly always shares its continuers with the base language, + so falling back to it is far better than falling back to English. + """ + candidates = [language] + if "-" in language: + candidates.append(language.split("-", 1)[0]) + if FALLBACK_LANGUAGE not in candidates: + candidates.append(FALLBACK_LANGUAGE) + return candidates + + +def load_phrases(language: str) -> CallerPhrases: + """Return the caller's phrase vocabulary for a language. + + Falls back to the base language and then to English rather than raising: a + missing translation should degrade the realism of the behavior, not abort the + run. The fallback is logged because English continuers in a non-English call + are a defect in the data, not an acceptable outcome. + """ + if language in _cache: + return _cache[language] + + data = _read_file() + for candidate in candidate_languages(language): + entry = data.get(candidate) + if not entry: + continue + if candidate != language: + logger.warning( + f"No caller phrases for {language!r}; falling back to {candidate!r}. " + f"Run scripts/add_culture_data.py --language {language} to generate them." + ) + phrases = CallerPhrases( + backchannels=list(entry.get("backchannels") or []), + barge_in_openers=list(entry.get("barge_in_openers") or []), + ) + _cache[language] = phrases + return phrases + + logger.error(f"No caller phrases for {language!r} and no {FALLBACK_LANGUAGE!r} fallback in {PHRASES_PATH}") + empty = CallerPhrases(backchannels=[], barge_in_openers=[]) + _cache[language] = empty + return empty diff --git a/src/eva/user_simulator/cascade/scheduler.py b/src/eva/user_simulator/cascade/scheduler.py new file mode 100644 index 00000000..8d96d57f --- /dev/null +++ b/src/eva/user_simulator/cascade/scheduler.py @@ -0,0 +1,166 @@ +"""Tick scheduler: virtual clock, playout queue, and turn-state machine.""" + +from __future__ import annotations + +from eva.user_simulator.cascade.adapter.base import Adapter +from eva.user_simulator.cascade.constants import ( + BYTES_PER_TICK, + LISTENER_CHECK_INTERVAL_MS, + WAIT_TO_RESPOND_OTHER_MS, + WAIT_TO_RESPOND_SELF_MS, + ms_to_ticks, +) +from eva.user_simulator.cascade.tick_result import TickResult, split_tick_audio + +_NEVER_SPOKE = 10**9 + + +class TickScheduler: + """Advances the virtual clock and decides who holds the floor. + + Caller turn boundaries are authored: an utterance is queued whole and drains + one tick at a time from a known start tick. The assistant's are detected from + consecutive silent ticks. Both land in the same synchronous step, so their + relative order can never come out ambiguous. This guarantee assumes a single + caller drives `run_tick` sequentially; concurrent calls are unsupported. + """ + + def __init__(self, adapter: Adapter, *, bytes_per_tick: int = BYTES_PER_TICK) -> None: + self._adapter = adapter + self._bytes_per_tick = bytes_per_tick + self._playout = bytearray() + self._tick = 0 + self._ticks_since_assistant_speech = _NEVER_SPOKE + self._ticks_since_caller_speech = _NEVER_SPOKE + self._assistant_has_spoken = False + self._awaiting_reply = False + self._caller_spoke_this_tick = False + self._backchannel_bytes = 0 + self._barge_in_armed = False + + @property + def tick(self) -> int: + """Current tick index; advances only after a successful `run_tick`.""" + return self._tick + + def enqueue_utterance(self, audio: bytes) -> None: + """Append audio to drain from the next tick onward. + + Appends concatenate into one continuous stream with no gap between them. + Queuing a logically separate utterance while one is still draining is the + caller's responsibility to avoid. + """ + self._playout.extend(audio) + + def enqueue_backchannel(self, audio: bytes) -> None: + """Append a continuer, which sounds but does not take the caller's turn. + + A backchannel earns no reply, so counting it as a turn leaves the caller + waiting for one that never comes and the call dies at the inactivity + timeout instead of reaching a goodbye. + """ + self._backchannel_bytes += len(audio) + self._playout.extend(audio) + + def arm_barge_in(self) -> None: + """Mark the next tick that puts caller audio on the wire as an interruption. + + Armed rather than passed directly because an interruption is enqueued as + audio and only reaches the wire on a later tick; the truncation must carry + the played position as of *that* tick, not as of the decision. + """ + self._barge_in_armed = True + + @property + def caller_is_speaking(self) -> bool: + """Whether caller audio is still queued for playout.""" + return bool(self._playout) + + @property + def assistant_has_spoken(self) -> bool: + """Whether the assistant has produced audio at any point in the call.""" + return self._assistant_has_spoken + + @property + def caller_spoke_this_tick(self) -> bool: + """Whether real caller audio went out on the most recent tick. + + Distinct from `caller_is_speaking`, which is true as soon as an utterance is + queued. This flips exactly on the ticks audio enters and leaves the wire, so + it dates the authored turn boundary rather than estimating it from silence. + """ + return self._caller_spoke_this_tick + + @property + def assistant_is_speaking(self) -> bool: + """Whether the assistant produced audio on the most recent tick.""" + return self._ticks_since_assistant_speech == 0 + + def is_check_tick(self) -> bool: + """Whether the listener-reaction checks should run now (tau: streaming.py:2514-2521).""" + if not self.assistant_is_speaking or self.caller_is_speaking: + return False + return self.tick % ms_to_ticks(LISTENER_CHECK_INTERVAL_MS) == 0 + + def may_take_turn(self) -> bool: + """Whether both silence thresholds are satisfied (tau: streaming.py:2590-2606). + + Gated on the assistant having spoken at least once: the assistant opens + the call with a greeting, and without this the caller would talk over it + on tick 0, since neither silence counter has anything to measure yet. + + Also gated on the assistant having replied since the caller's last turn: + the silence thresholds alone are satisfied a fixed time after the caller + stops talking regardless of whether a reply ever arrived, which lets the + caller repeat itself into a slow assistant. An assistant that stops replying + altogether is an inactivity timeout, which the simulator ends the call on. + """ + if not self._assistant_has_spoken or self._awaiting_reply: + return False + return self._ticks_since_assistant_speech > ms_to_ticks( + WAIT_TO_RESPOND_OTHER_MS + ) and self._ticks_since_caller_speech > ms_to_ticks(WAIT_TO_RESPOND_SELF_MS) + + async def run_tick(self) -> TickResult: + """Exchange one tick with the adapter and advance the turn-state machine. + + The playout queue is only drained after the adapter call succeeds, so a + raised exception leaves the queue and tick count exactly as they were. + """ + outgoing, consumed = self._peek_chunk() + barge_in = self._barge_in_armed and outgoing is not None + result = await self._adapter.run_tick(self._tick, outgoing, barge_in=barge_in) + del self._playout[:consumed] + if barge_in: + self._barge_in_armed = False + + # Backchannel bytes sit at the head of the queue, so this tick is a continuer + # only while they remain. Anything past them is real speech and takes the turn. + was_backchannel = consumed > 0 and self._backchannel_bytes > 0 + self._backchannel_bytes = max(0, self._backchannel_bytes - consumed) + + self._caller_spoke_this_tick = outgoing is not None + self._ticks_since_caller_speech = 0 if outgoing else self._ticks_since_caller_speech + 1 + self._ticks_since_assistant_speech = ( + 0 if result.has_assistant_speech else self._ticks_since_assistant_speech + 1 + ) + self._assistant_has_spoken = self._assistant_has_spoken or result.has_assistant_speech + if result.has_assistant_speech: + self._awaiting_reply = False + if outgoing and not was_backchannel: + self._awaiting_reply = True + + self._tick += 1 + return result + + def _peek_chunk(self) -> tuple[bytes | None, int]: + """Preview one tick of queued caller audio without consuming it. + + Returns the padded chunk (or None when silent) and how many raw bytes + of `_playout` it was drawn from, for the caller to commit after success. + """ + if not self._playout: + return None, 0 + raw = bytes(self._playout[: self._bytes_per_tick]) + chunk, _ = split_tick_audio(raw, self._bytes_per_tick) + return chunk, len(raw) diff --git a/src/eva/user_simulator/cascade/simulator.py b/src/eva/user_simulator/cascade/simulator.py new file mode 100644 index 00000000..b650f8ef --- /dev/null +++ b/src/eva/user_simulator/cascade/simulator.py @@ -0,0 +1,749 @@ +"""Self-hosted STT/LLM/TTS caller driven by the tick scheduler.""" + +from __future__ import annotations + +import audioop +import re +import time +from pathlib import Path + +import websockets + +from eva.assistant.services.llm import LiteLLMClient +from eva.models.config import CascadeSimulatorConfig, PerturbationConfig +from eva.user_simulator.base import AbstractUserSimulator +from eva.user_simulator.cascade.adapter.base import Adapter +from eva.user_simulator.cascade.adapter.realtime_ws import RealtimeWSAdapter +from eva.user_simulator.cascade.adapter.tick_driven import MAX_INACTIVE_SECONDS, TickDrivenAdapter +from eva.user_simulator.cascade.constants import ( + CALLER_SAMPLE_RATE, + INACTIVITY_TIMEOUT_MS, + TICK_DURATION_MS, + TRANSCRIPT_WAIT_MS, + WAIT_TO_RESPOND_OTHER_MS, + ms_to_ticks, +) +from eva.user_simulator.cascade.decision_log import DecisionLog +from eva.user_simulator.cascade.decisions import ListenerDecisions, parse_yes_no +from eva.user_simulator.cascade.phrase_cache import PhraseCache +from eva.user_simulator.cascade.phrases import load_phrases +from eva.user_simulator.cascade.scheduler import TickScheduler +from eva.user_simulator.cascade.stt_livekit import LiveKitStreamingSTT +from eva.user_simulator.cascade.tick_result import TickResult +from eva.user_simulator.cascade.tts import CartesiaTTS + +# Shared with the OpenAI Realtime provider so both simulators hang up on the same rules. +from eva.user_simulator.openai_realtime import END_CALL_DESCRIPTION +from eva.utils.logging import get_logger +from eva.utils.prompt_manager import PromptManager + +logger = get_logger(__name__) + +_FENCE = re.compile(r"^```[a-z]*\s*|\s*```$", re.MULTILINE) + +END_CALL_TOOL = { + "type": "function", + "function": { + "name": "end_call", + "description": END_CALL_DESCRIPTION, + "parameters": {"type": "object", "properties": {}, "required": []}, + }, +} + + +def parse_turn_response(raw: str) -> str: + """Return the spoken line from the model's reply, stripping any stray code fence.""" + return _FENCE.sub("", raw).strip() + + +def _flip_role(role: str) -> str: + """Swap user/assistant so the caller LLM sees its own lines tagged assistant.""" + return "assistant" if role == "user" else "user" + + +def extract_turn(message: object) -> tuple[str, bool]: + """Read (utterance, end_call) from whatever LiteLLMClient returned. + + ``complete()`` returns a bare ``str`` when the model made no tool call and a + message object when it did, so both shapes must be handled. + """ + if isinstance(message, str): + return parse_turn_response(message), False + content = getattr(message, "content", None) or "" + calls = getattr(message, "tool_calls", None) or [] + end_call = any(getattr(call.function, "name", "") == "end_call" for call in calls) + return parse_turn_response(content), end_call + + +def extract_optional_line(message: object) -> str: + """Read a bare spoken line from a dedicated call, empty when the model declined. + + A bare line rather than a JSON field: demanding JSON on a caller call suppressed + the end_call tool call entirely (Plan 1). + """ + content = message if isinstance(message, str) else (getattr(message, "content", None) or "") + line = _FENCE.sub("", content).strip() + return "" if line.upper() == "NONE" else line + + +def interrupt_slip_ms(*, elapsed_s: float) -> int: + """How far past its intended moment a barge-in actually landed. + + Measured on the wall clock, not the tick counter. `run_tick` is only pumped by + the `_run` loop, so while `_play_interruption` awaits its generation no tick can + advance — a tick-delta slip is structurally always zero and the staleness check + built on it never engages. + """ + return max(0, int(elapsed_s * 1000)) + + +def summarize_goal(goal: dict) -> str: + """State what the caller wants and what would end the call, for the listener checks. + + Field meanings live in the prompt template rather than here, so the explanations stay + static across the many calls this decision makes rather than being rebuilt per call. + `edge_cases` and `information_required` are deliberately omitted: they are long and + describe how to answer questions, not whether the goal is finished. + """ + tree = goal.get("decision_tree", {}) or {} + sections = [ + ("GOAL", goal.get("high_level_user_goal")), + ("MUST HAVE", tree.get("must_have_criteria")), + ("NICE TO HAVE", tree.get("nice_to_have_criteria")), + ("HOW THEY EVALUATE OPTIONS", tree.get("negotiation_behavior")), + ("RESOLVED WHEN", tree.get("resolution_condition")), + ("FAILED WHEN", tree.get("failure_condition")), + ("ESCALATION", tree.get("escalation_behavior")), + ] + lines: list[str] = [] + for label, value in sections: + if not value: + continue + if isinstance(value, list): + lines.append(f"{label}:") + lines += [f"- {item}" for item in value] + else: + lines.append(f"{label}: {value}") + return "\n".join(lines) + + +def is_new_assistant_turn(*, ticks_silent_before: int) -> bool: + """Whether assistant audio arriving now starts a new turn or resumes the current one. + + A pause shorter than the turn-end threshold is a gap *inside* one turn — between + sentences, or a hole in the audio stream — not a new turn. Treating any single quiet + tick as a boundary re-armed the interruption cap mid-utterance, so one assistant turn + could collect several barge-ins. + """ + return ticks_silent_before >= ms_to_ticks(WAIT_TO_RESPOND_OTHER_MS) + + +def should_drop_interrupt(*, assistant_still_speaking: bool, same_assistant_turn: bool) -> bool: + """Whether a barge-in has gone stale and should be abandoned. + + Staleness is a fact about the assistant, not about how long generation took: a line is + still a real interruption whenever the assistant is mid-utterance, however many wall-clock + seconds elapsed. Both conditions are needed — "still speaking" alone is satisfied by a + *later* turn, which would land the line as a non-sequitur against speech it never heard. + """ + return not (assistant_still_speaking and same_assistant_turn) + + +async def candidate_is_relevant(llm, *, candidate: str, heard: str) -> bool: + """Whether a pre-generated interruption still fits what the assistant is saying. + + Fails closed: an unusable answer means we fall back to generating fresh + content rather than firing something that may have gone stale. + """ + prompt = PromptManager().get_prompt("user_simulator.cascade_relevance_gate", candidate=candidate, heard=heard) + try: + reply = await llm.decide(prompt) + except Exception as exc: + logger.warning(f"Relevance gate failed, discarding candidate: {exc}") + return False + return parse_yes_no(reply) + + +def play_backchannel(scheduler, cache, phrases: list[str]) -> str: + """Queue a cached continuer and return the phrase that was chosen. + + Queued as a backchannel, not an utterance: it must not consume the caller's + turn, or the caller waits for a reply the continuer never earns. + """ + phrase = cache.choose(phrases) + scheduler.enqueue_backchannel(cache.get(phrase)) + return phrase + + +class _DecisionClient: + """Adapts LiteLLMClient to the single-prompt interface the checks expect.""" + + def __init__(self, client: LiteLLMClient) -> None: + self._client = client + + async def decide(self, prompt: str) -> str: + """Ask one YES/NO question and return the raw reply.""" + message, _stats = await self._client.complete(messages=[{"role": "user", "content": prompt}]) + if isinstance(message, str): + return message + return getattr(message, "content", None) or "" + + +TICK_DRIVEN_FRAMEWORKS = frozenset({"openai_realtime"}) +"""Frameworks whose clock the caller can own. Others keep real-time streaming.""" + + +def adapter_class_for_framework(framework: str) -> type[Adapter]: + """Pick the adapter for a framework, defaulting to real-time streaming. + + Defaulting to real-time is deliberate: it works everywhere, whereas + tick-driving requires the assistant to have no wall-clock timers of its own. + """ + if framework in TICK_DRIVEN_FRAMEWORKS: + return TickDrivenAdapter + return RealtimeWSAdapter + + +class CascadeUserSimulator(AbstractUserSimulator): + """Simulated caller built from independently chosen STT, LLM, and TTS models.""" + + _ticks_awaiting_transcript = 0 + _ticks_assistant_silent = 0 + _ticks_since_assistant_started = 0 + _may_interrupt_this_turn = False + _assistant_turn_index = 0 + _last_checked_text = "" + + def __init__( + self, + current_date_time: str, + persona_config: dict, + goal: dict, + server_url: str, + output_dir: Path, + agent_id: str, + timeout: int = 600, + perturbation_config: PerturbationConfig | None = None, + language: str = "en", + *, + simulator_config: CascadeSimulatorConfig, + framework: str = "pipecat", + ) -> None: + super().__init__( + current_date_time=current_date_time, + persona_config=persona_config, + goal=goal, + server_url=server_url, + output_dir=output_dir, + agent_id=agent_id, + timeout=timeout, + perturbation_config=perturbation_config, + language=language, + provider="cascade", + ) + self._config = simulator_config + self._framework = framework + self._stt = LiveKitStreamingSTT(simulator_config.stt, simulator_config.stt_params, language=language) + self._tts = CartesiaTTS(simulator_config.tts_params, language=language) + self._llm = LiteLLMClient(model=simulator_config.llm) + self._voice_id = self._tts.voice_for_persona(persona_config) + # The caller's out-of-turn vocabulary is language data, not code. + self._phrases = load_phrases(language) + self._history: list[dict[str, str]] = [] + # Shared by the listener checks and the relevance gate so both cost one client. + self._decision_client = _DecisionClient(LiteLLMClient(model=simulator_config.decision_llm)) + self._phrase_cache: PhraseCache | None = None + self._decisions: ListenerDecisions | None = None + self._candidate_text = "" + self._candidate_audio = b"" + self._decision_log = DecisionLog(self.output_dir / "user_simulator_decisions.jsonl") + + async def run_conversation(self) -> str: + """Run the tick loop until the call ends, and return the end reason.""" + try: + await self._run() + except Exception as exc: + logger.exception(f"Cascade simulator failed: {exc}") + self._end_reason = "error" + self.event_logger.log_error(str(exc)) + finally: + self._save_clean_user_audio(CALLER_SAMPLE_RATE) + self.event_logger.save() + self._decision_log.save() + logger.info(f"Caller decision trace: {self._decision_log.summary()}") + return self._end_reason + + async def _run(self) -> None: + """Drive the scheduler until end_call, timeout, or disconnect.""" + websocket = await websockets.connect(self.server_url) + adapter_cls = adapter_class_for_framework(self._framework) + adapter = adapter_cls( + websocket=websocket, + conversation_id=self._record_id or "cascade", + perturbator=self._perturbator, + ) + scheduler = TickScheduler(adapter) + + await adapter.start() + await self._stt.start() + await self._prepare_listener_behaviors() + self.event_logger.log_connection_state("connected", {"server_url": self.server_url}) + + max_ticks = self.timeout * 1000 // TICK_DURATION_MS + assistant_was_speaking = False + caller_was_speaking = False + try: + while scheduler.tick < max_ticks and not self._conversation_done.is_set(): + result = await scheduler.run_tick() + # Fed on every tick, speech or silence, so Scribe sees a continuous stream + # and never idles out; committed exactly on the speech->silence transition, + # which is what closes the utterance so take_committed() below isn't starved. + commit = assistant_was_speaking and not result.has_assistant_speech + if result.has_assistant_speech != assistant_was_speaking: + logger.debug( + f"tick {scheduler.tick}: assistant speech " + f"{'started' if result.has_assistant_speech else 'ended'} " + f"(raw={result.assistant_audio_raw_bytes}B)" + ) + await self._stt.feed(result.assistant_audio, commit=commit) + self._log_audio_boundaries(scheduler, result, assistant_was_speaking, caller_was_speaking) + caller_was_speaking = scheduler.caller_spoke_this_tick + assistant_was_speaking = result.has_assistant_speech + if result.provider_stalled: + # Distinct from inactivity_timeout, which is a *legitimate* end the + # metrics treat as definitive when the user spoke last. A stall is a + # dead peer: the record is invalid and the runner should retry it, + # which is what the reason not being "goodbye" already means to it. + logger.error( + f"tick {scheduler.tick}: no assistant audio for " + f"{MAX_INACTIVE_SECONDS}s; abandoning the conversation as unusable" + ) + self._on_conversation_end("provider_stalled") + break + # Captured before the inactivity check, which clears it on a speech tick. + silent_before = self._ticks_assistant_silent + if self._assistant_is_inactive(scheduler, result): + logger.warning( + f"tick {scheduler.tick}: assistant silent for " + f"{INACTIVITY_TIMEOUT_MS // 1000}s; ending the conversation" + ) + self._on_conversation_end("inactivity_timeout") + break + if result.has_assistant_speech: + if is_new_assistant_turn(ticks_silent_before=silent_before): + self._ticks_since_assistant_started = 0 + # One roll per assistant turn, so a turn carries at most one barge-in. + self._may_interrupt_this_turn = self._config.enable_interruptions + self._assistant_turn_index += 1 + # The transcript is consumed between turns, so it can shrink back + # toward the in-flight partial and collide with a value already + # judged. Clearing here keeps "unchanged" meaning "unchanged + # within this turn", which is the only span it is asked about. + self._last_checked_text = "" + self._decision_log.log( + "assistant_turn_start", + tick=scheduler.tick, + turn_index=self._assistant_turn_index, + ticks_silent_before=silent_before, + armed_interrupt=self._may_interrupt_this_turn, + ) + self._ticks_since_assistant_started += 1 + is_check = scheduler.is_check_tick() + self._log_tick_state(scheduler, result, is_check_tick=is_check) + if is_check and await self._run_checks(scheduler): + break + continue + self._log_tick_state(scheduler, result, is_check_tick=False) + if scheduler.caller_is_speaking or not scheduler.may_take_turn(): + continue + heard, waiting = self._collect_heard_text(scheduler) + if waiting: + continue + if await self._take_turn(scheduler, heard): + break + else: + if not self._conversation_done.is_set(): + self._on_conversation_end("timeout") + finally: + await self._stt.stop() + await adapter.stop() + self.event_logger.log_connection_state("session_ended", {"reason": self._end_reason}) + + async def _prepare_listener_behaviors(self) -> None: + """Pre-render the fixed vocabularies and build the checks, when any is enabled. + + Rendering happens once at connect time rather than on demand: a cached phrase + is the only way a 300ms reaction lands on the tick its check chose. + """ + vocabulary: list[str] = [] + if self._config.enable_backchannel: + vocabulary += self._phrases.backchannels + if self._config.enable_interruptions: + vocabulary += self._phrases.barge_in_openers + if not vocabulary: + return + + self._phrase_cache = PhraseCache(self._tts, voice_id=self._voice_id) + await self._phrase_cache.prerender(vocabulary) + prompts = PromptManager() + self._decisions = ListenerDecisions( + self._decision_client, + interrupt_prompt=prompts.get_template("user_simulator.interruption_decision"), + backchannel_prompt=prompts.get_template("user_simulator.backchannel_decision"), + user_goal=summarize_goal(self.goal), + ) + + async def _run_checks(self, scheduler: TickScheduler) -> bool: + """Run the listener-reaction checks and act on the verdict. True means hang up. + + Gated on there being something new to judge. The checks fire on a timer while + the assistant speaks — ~101 times in one measured call, two concurrent LLM calls + each — but their only input is the transcript, so a tick where the transcript has + not moved re-asks a question already answered. Skipping those is free: identical + input, identical verdict. + """ + if self._decisions is None or self._phrase_cache is None: + return False + + history = self._stt.buffer.current_text() + # Nothing transcribed yet is not a question worth asking, and a transcript that + # has not moved re-asks one already answered. + if not history or history == self._last_checked_text: + return False + self._last_checked_text = history + + # Backchannelling stays available after a barge-in: a listener who cuts in and + # later hums along is ordinary, and the one-barge-in-per-turn cap is about not + # talking over the assistant repeatedly, not about going quiet afterwards. + # `allow_interrupt` alone already skips the interrupt call without reaching the + # model once the cap is spent. + verdict = await self._decisions.evaluate( + history, + allow_interrupt=self._may_interrupt_this_turn, + allow_backchannel=self._config.enable_backchannel, + ) + self._decision_log.log( + "listener_check", + tick=scheduler.tick, + allow_interrupt=self._may_interrupt_this_turn, + allow_backchannel=self._config.enable_backchannel, + heard_chars=len(history), + heard=history, + interrupt_ran=verdict.interrupt_trace.ran, + interrupt_raw=verdict.interrupt_trace.raw, + interrupt_latency_ms=verdict.interrupt_trace.latency_ms, + interrupt_error=verdict.interrupt_trace.error, + backchannel_ran=verdict.backchannel_trace.ran, + backchannel_raw=verdict.backchannel_trace.raw, + backchannel_error=verdict.backchannel_trace.error, + should_interrupt=verdict.should_interrupt, + should_backchannel=verdict.should_backchannel, + ) + if verdict.should_interrupt: + self._may_interrupt_this_turn = False + return await self._play_interruption(scheduler) + if verdict.should_backchannel: + phrase = play_backchannel(scheduler, self._phrase_cache, self._phrases.backchannels) + # Recorded too, or the saved clean track diverges from what went on the wire. + self._record_audio("user_clean", self._phrase_cache.get(phrase)) + self.event_logger.log_event("backchannel", {"text": phrase, "tick_index": scheduler.tick}) + return False + + async def _play_interruption(self, scheduler: TickScheduler) -> bool: + """Decide what to say, then voice the opener and the content together. + + Returns True when the caller decided to hang up. + + The content is generated *before* anything reaches the wire. Emitting the opener + first hid its ~1s latency, but committed the caller to barging in before knowing + whether it had anything to say: a hang-up or a dropped-as-stale line then left an + orphaned "Actually—" hanging with nothing behind it, which is worse than either a + late line or silence. Generating first means "say nothing" is actually available. + """ + if self._phrase_cache is None: + return False + intended_tick = scheduler.tick + intended_turn = self._assistant_turn_index + started_at = time.monotonic() + opener = self._phrase_cache.choose(self._phrases.barge_in_openers) + opener_audio = self._phrase_cache.get(opener) + + if self._config.speculative_generation and self._candidate_audio: + candidate, audio = self._candidate_text, self._candidate_audio + self._candidate_text, self._candidate_audio = "", b"" + relevant = await candidate_is_relevant( + self._decision_client, candidate=candidate, heard=self._stt.buffer.current_text() + ) + self._decision_log.log("relevance_gate", tick=intended_tick, candidate=candidate, relevant=relevant) + if relevant: + # Tell the adapter the next tick that reaches the wire cuts the + # assistant off, so a tick-driven transport can truncate the audio + # the caller never heard. Ignored on the real-time path. + scheduler.arm_barge_in() + scheduler.enqueue_utterance(opener_audio) + self._record_audio("user_clean", opener_audio) + scheduler.enqueue_utterance(audio) + self._record_audio("user_clean", audio) + self._history.append({"role": "user", "content": candidate}) + self._on_user_speaks(candidate) + self.event_logger.log_event( + "interruption", + { + "text": candidate, + "opener": opener, + "intended_tick": intended_tick, + "actual_tick": scheduler.tick, + "slip_ms": interrupt_slip_ms(elapsed_s=time.monotonic() - started_at), + "speculative": True, + "dropped": False, + }, + ) + self._decision_log.log( + "interruption", tick=intended_tick, outcome="spoken", speculative=True, text=candidate + ) + return False + self.event_logger.log_event("interruption_candidate_rejected", {"text": candidate}) + self._decision_log.log("interruption", tick=intended_tick, outcome="candidate_rejected", text=candidate) + + # Consumed, not peeked: leaving it in the buffer would re-append the same + # assistant prefix at the next ordinary turn and duplicate it in the history. + heard = self._stt.buffer.heard_text() + self._stt.buffer.take_committed() + self._stt.buffer.in_flight = "" + if heard: + self._history.append({"role": "assistant", "content": heard}) + self._on_assistant_speaks(heard) + message, _stats = await self._llm.complete(messages=self._messages(), tools=[END_CALL_TOOL]) + utterance, end_call = extract_turn(message) + + slip = interrupt_slip_ms(elapsed_s=time.monotonic() - started_at) + # A hang-up is never stale: the caller has decided the call is over, and + # dropping it here is what left conversations looping until the timeout. + dropped = not end_call and should_drop_interrupt( + assistant_still_speaking=scheduler.assistant_is_speaking, + same_assistant_turn=self._assistant_turn_index == intended_turn, + ) + self.event_logger.log_event( + "interruption", + { + "text": utterance, + "opener": opener, + "intended_tick": intended_tick, + "actual_tick": scheduler.tick, + "slip_ms": slip, + "dropped": dropped, + "end_call": end_call, + }, + ) + self._decision_log.log( + "interruption", + tick=intended_tick, + outcome="dropped" if dropped else ("end_call" if end_call else "spoken"), + text=utterance, + slip_ms=slip, + assistant_still_speaking=scheduler.assistant_is_speaking, + intended_turn=intended_turn, + actual_turn=self._assistant_turn_index, + ) + # Nothing has reached the wire yet, so a stale line or a hang-up costs no audio. + if dropped: + return False + + if utterance: + scheduler.arm_barge_in() + scheduler.enqueue_utterance(opener_audio) + self._record_audio("user_clean", opener_audio) + self._history.append({"role": "user", "content": utterance}) + self._on_user_speaks(utterance) + async for chunk in self._tts.stream(utterance, voice_id=self._voice_id): + self._record_audio("user_clean", chunk) + scheduler.enqueue_utterance(chunk) + + if end_call: + self._on_conversation_end("goodbye") + return end_call + + def _assistant_is_inactive(self, scheduler: TickScheduler, result: TickResult) -> bool: + """Whether the assistant has produced no audio for INACTIVITY_TIMEOUT_MS *contiguously*. + + Mirrors ElevenLabsUserSimulator's keep-alive rule so both providers record the + same terminal state: conversation_valid_end treats inactivity_timeout with the + user speaking last as a definitive end, not a failure. + + Must be called on every tick, speech or silence. It was previously reached only on + silent ticks, which made the reset below dead code: the counter then measured + *cumulative* silence over the whole call and killed healthy conversations once their + quiet ticks happened to total two minutes. + """ + if result.has_assistant_speech: + self._ticks_assistant_silent = 0 + return False + self._ticks_assistant_silent += 1 + return scheduler.assistant_has_spoken and self._ticks_assistant_silent > ms_to_ticks(INACTIVITY_TIMEOUT_MS) + + def _log_audio_boundaries( + self, + scheduler: TickScheduler, + result: TickResult, + assistant_was_speaking: bool, + caller_was_speaking: bool, + ) -> None: + """Emit audio_start/audio_end for both roles, which is how metrics number turns. + + The caller's boundaries are authored rather than detected: the playout queue + drains on a known tick, so these stamp the real edges instead of a + silence-threshold estimate that has to be back-dated (see + BotToBotAudioBridge, whose end detection lags by ~600ms). + """ + seconds = result.wall_clock_ms / 1000 + caller_speaking = scheduler.caller_spoke_this_tick + if caller_speaking and not caller_was_speaking: + self.event_logger.log_audio_start("simulated_user", seconds) + elif not caller_speaking and caller_was_speaking: + self.event_logger.log_audio_end("simulated_user", seconds) + if result.has_assistant_speech and not assistant_was_speaking: + self.event_logger.log_audio_start("assistant", seconds) + elif not result.has_assistant_speech and assistant_was_speaking: + self.event_logger.log_audio_end("assistant", seconds) + + def _log_tick_state(self, scheduler: TickScheduler, result: TickResult, *, is_check_tick: bool) -> None: + """Trace one tick's speech state, so a check that never ran can be traced to its gate. + + `rms` is why this exists: `has_assistant_speech` is true for any non-zero bytes, + digital silence included, so a transport that pads with silence reads as continuous + speech. Recording both lets that be measured rather than inferred. + + `transcript_moved` serves the same purpose for the other gate. A check tick only + reaches the judges when the transcript has changed since the last one judged, so + without this a check tick with no following `listener_check` is ambiguous between + "nothing new was heard" and "the check ran and something went wrong". + """ + raw = result.assistant_audio[: result.assistant_audio_raw_bytes] + history = self._stt.buffer.current_text() + self._decision_log.log( + "tick", + tick=scheduler.tick, + has_assistant_speech=result.has_assistant_speech, + raw_bytes=result.assistant_audio_raw_bytes, + rms=audioop.rms(raw, 2) if len(raw) >= 2 else 0, + caller_is_speaking=scheduler.caller_is_speaking, + caller_spoke_this_tick=scheduler.caller_spoke_this_tick, + ticks_assistant_silent=self._ticks_assistant_silent, + ticks_since_assistant_started=self._ticks_since_assistant_started, + may_interrupt_this_turn=self._may_interrupt_this_turn, + is_check_tick=is_check_tick, + transcript_moved=bool(history) and history != self._last_checked_text, + has_candidate=bool(self._candidate_audio), + ) + + def _collect_heard_text(self, scheduler: TickScheduler) -> tuple[str, bool]: + """Return what the assistant said and whether to keep waiting for it. + + Finalization is not instantaneous, so an empty buffer at the first turn + opportunity usually means "not ready yet" rather than "nothing was said". + Retrying on later ticks is the wait; the in-flight partial is the fallback + once that budget is spent. + """ + heard = self._stt.buffer.take_committed() + if heard: + self._ticks_awaiting_transcript = 0 + return heard, False + + self._ticks_awaiting_transcript += 1 + if self._ticks_awaiting_transcript <= ms_to_ticks(TRANSCRIPT_WAIT_MS): + return "", True + + partial = self._stt.buffer.in_flight + self._stt.buffer.in_flight = "" + if partial: + logger.warning( + f"tick {scheduler.tick}: no final transcript after {TRANSCRIPT_WAIT_MS}ms; " + f"falling back to the in-flight partial: {partial[:120]!r}" + ) + self.event_logger.log_event("transcript_partial_fallback", {"text": partial, "tick_index": scheduler.tick}) + return partial, False + + # Nothing was heard at all. Keep waiting rather than speaking into the void: + # an assistant that never replies is an inactivity timeout, handled in _run. + return "", True + + async def _take_turn(self, scheduler: TickScheduler, heard: str) -> bool: + """Generate, synthesize, and queue one caller turn. Returns True to hang up.""" + if heard: + self._history.append({"role": "assistant", "content": heard}) + self._on_assistant_speaks(heard) + + message, _stats = await self._llm.complete(messages=self._messages(), tools=[END_CALL_TOOL]) + utterance, end_call = extract_turn(message) + + if utterance: + self._history.append({"role": "user", "content": utterance}) + self._on_user_speaks(utterance) + audio = await self._tts.synthesize(utterance, voice_id=self._voice_id) + self._record_audio("user_clean", audio) + scheduler.enqueue_utterance(audio) + self.event_logger.log_event("caller_turn", {"text": utterance, "tick_index": scheduler.tick}) + + if end_call: + self._on_conversation_end("goodbye") + return True + + # After the audio is queued: the caller is now speaking, so this generation + # runs behind its own outgoing audio and costs no conversational latency. + if utterance: + await self._prerender_candidate(utterance) + return False + + async def _prerender_candidate(self, utterance: str) -> None: + """Pre-generate and pre-render the line the caller would barge in with. + + Done on the caller's own turn, where the latency is already hidden behind + its outgoing audio, so a later barge-in can fire without waiting on + generation. The relevance gate is what keeps this from degrading into a + scripted interruption that lands as a non-sequitur. + """ + self._candidate_text, self._candidate_audio = "", b"" + # A candidate only ever gets spoken by a barge-in, so rendering one with + # interruptions off buys an LLM call and a synthesis per turn for nothing. + if not self._config.speculative_generation or not self._config.enable_interruptions: + return + started = time.monotonic() + prompt = PromptManager().get_prompt("user_simulator.cascade_next_interruption", utterance=utterance) + try: + message, _stats = await self._llm.complete( + messages=[*self._messages(), {"role": "user", "content": prompt}] + ) + except Exception as exc: + logger.warning(f"Speculative interruption generation failed: {exc}") + self._decision_log.log("candidate_generation", ok=False, error=str(exc)) + return + candidate = extract_optional_line(message) + if not candidate: + raw = message if isinstance(message, str) else (getattr(message, "content", None) or "") + self._decision_log.log("candidate_generation", ok=False, declined=True, after=utterance, raw=raw[:400]) + return + self._candidate_text = candidate + self._candidate_audio = await self._tts.synthesize(candidate, voice_id=self._voice_id) + self._decision_log.log( + "candidate_generation", + ok=True, + after=utterance, + candidate=candidate, + audio_bytes=len(self._candidate_audio), + latency_ms=int((time.monotonic() - started) * 1000), + ) + + def _messages(self) -> list[dict[str, str]]: + """Build the message list: the shared per-domain caller prompt plus flipped history. + + The system prompt is `_build_prompt()` unmodified — the same per-domain prompt the other + providers use, which already carries the persona, goal, and end_call rules. + + `self._history` is kept in conversation-truth roles (assistant said by the agent, user + said by the caller) since it also feeds logging. This LLM is itself the assistant + in its own frame, so that history must be flipped here or a message tagged "assistant" + reads to the model as its own prior output and it echoes it back. + """ + messages = [{"role": "system", "content": self._build_prompt()}] + messages += [{"role": _flip_role(turn["role"]), "content": turn["content"]} for turn in self._history] + return messages diff --git a/src/eva/user_simulator/cascade/stt.py b/src/eva/user_simulator/cascade/stt.py new file mode 100644 index 00000000..750136fe --- /dev/null +++ b/src/eva/user_simulator/cascade/stt.py @@ -0,0 +1,41 @@ +"""Transcript accumulation for the caller's speech-to-text.""" + +from __future__ import annotations + + +class TranscriptBuffer: + """Accumulates committed transcript segments separately from the in-flight partial.""" + + def __init__(self) -> None: + self.committed = "" + self.in_flight = "" + + def apply_partial(self, text: str) -> None: + """Replace the in-flight partial with the latest partial transcript.""" + self.in_flight = text + + def commit(self, text: str) -> None: + """Append a finalized segment to the committed text and clear the partial.""" + self.committed = f"{self.committed} {text}".strip() if self.committed else text + self.in_flight = "" + + def current_text(self) -> str: + """Return everything heard so far, marking the partial as still in progress. + + The marker is load-bearing: the backchannel prompt's few-shot examples all + end in it, and it is what tells the check to judge only the complete + sentences rather than the truncated trailing word. + """ + if not self.in_flight: + return self.committed + return f"{self.committed} {self.in_flight} [CURRENTLY SPEAKING, INCOMPLETE]".strip() + + def heard_text(self) -> str: + """Return everything heard so far without the in-progress marker, for transcript use.""" + return f"{self.committed} {self.in_flight}".strip() + + def take_committed(self) -> str: + """Return and clear the committed text.""" + text = self.committed + self.committed = "" + return text diff --git a/src/eva/user_simulator/cascade/stt_livekit.py b/src/eva/user_simulator/cascade/stt_livekit.py new file mode 100644 index 00000000..50a2964b --- /dev/null +++ b/src/eva/user_simulator/cascade/stt_livekit.py @@ -0,0 +1,117 @@ +"""Streaming caller STT backed by LiveKit Agents plugins, used without a room.""" + +from __future__ import annotations + +import asyncio +import contextlib +import os +from typing import Any + +from eva.user_simulator.cascade.constants import CALLER_SAMPLE_RATE +from eva.user_simulator.cascade.stt import TranscriptBuffer +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + +DEFAULT_MODELS = {"elevenlabs": "scribe_v2_realtime"} +API_KEY_ENV = {"elevenlabs": "ELEVENLABS_API_KEY"} + + +def build_livekit_stt(provider: str, params: dict[str, Any]) -> Any: + """Construct the LiveKit plugin STT for a provider, without provider-side endpointing.""" + model = params.get("model") or DEFAULT_MODELS.get(provider) + api_key = params.get("api_key") or os.environ.get(API_KEY_ENV.get(provider, ""), "") + if provider == "elevenlabs": + from livekit.plugins import elevenlabs + + # server_vad is deliberately not passed: the plugin then selects + # commit_strategy=manual, leaving the turn boundary ours to decide. + return elevenlabs.STT(model=model, api_key=api_key) + raise ValueError(f"Unsupported caller STT provider: {provider!r}. Supported: {sorted(DEFAULT_MODELS)}") + + +class LiveKitStreamingSTT: + """Transcribes assistant audio via a LiveKit STT plugin driven by our tick clock. + + Finalization is ours: `feed(..., commit=True)` maps to the plugin's `flush()` + sentinel rather than waiting for provider endpointing, which keeps the tick + scheduler the only thing deciding where a turn ends. + """ + + def __init__(self, provider: str, params: dict[str, Any], *, language: str = "en") -> None: + self.provider = provider + self.params = dict(params) + self.model = params.get("model") or DEFAULT_MODELS.get(provider, "") + self.language = language + self.buffer = TranscriptBuffer() + self._stt: Any = None + self._stream: Any = None + self._reader: asyncio.Task | None = None + self._http: Any = None + + async def start(self) -> None: + """Open the plugin's HTTP context and streaming recognizer.""" + from livekit.agents.utils import http_context + + # Plugins outside the agent worker have no ambient session; without this they + # raise "http session outside of a job context" on first use. + self._http = http_context.open() + await self._http.__aenter__() + self._stt = build_livekit_stt(self.provider, self.params) + self._stream = self._stt.stream() + self._reader = asyncio.create_task(self._receive_loop()) + + async def feed(self, pcm: bytes, *, commit: bool = False) -> None: + """Push one tick of assistant audio, optionally closing the utterance.""" + if self._stream is None: + return + from livekit import rtc + + try: + self._stream.push_frame( + rtc.AudioFrame( + data=pcm, + sample_rate=CALLER_SAMPLE_RATE, + num_channels=1, + samples_per_channel=len(pcm) // 2, + ) + ) + if commit: + self._stream.flush() + except Exception as exc: + logger.warning(f"LiveKit STT feed failed: {exc}") + + async def stop(self) -> None: + """Close the recognizer and HTTP context. Safe to call twice.""" + if self._reader is not None: + self._reader.cancel() + with contextlib.suppress(asyncio.CancelledError): + await self._reader + self._reader = None + if self._stream is not None: + with contextlib.suppress(Exception): + await self._stream.aclose() + self._stream = None + if self._http is not None: + with contextlib.suppress(Exception): + await self._http.__aexit__(None, None, None) + self._http = None + + async def _receive_loop(self) -> None: + """Fold interim and final transcripts into the buffer as the plugin emits them.""" + from livekit.agents import stt as lk_stt + + try: + async for event in self._stream: + alternatives = getattr(event, "alternatives", None) or [] + text = alternatives[0].text if alternatives else "" + if not text: + continue + if event.type == lk_stt.SpeechEventType.INTERIM_TRANSCRIPT: + self.buffer.apply_partial(text) + elif event.type == lk_stt.SpeechEventType.FINAL_TRANSCRIPT: + self.buffer.commit(text) + except asyncio.CancelledError: + raise + except Exception: + logger.exception("LiveKit STT receive loop failed; caller may stop hearing the assistant") diff --git a/src/eva/user_simulator/cascade/tick_result.py b/src/eva/user_simulator/cascade/tick_result.py new file mode 100644 index 00000000..f2343459 --- /dev/null +++ b/src/eva/user_simulator/cascade/tick_result.py @@ -0,0 +1,72 @@ +"""Per-tick exchange record between the scheduler and an adapter.""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import NamedTuple + +from eva.user_simulator.cascade.constants import BYTES_PER_TICK, SILENCE_BYTE, TICK_DURATION_MS + + +@dataclass(frozen=True) +class TickResult: + """What one tick of the conversation produced.""" + + tick_number: int + """Monotonic simulation-clock index; the ordering key for events, unlike wall_clock_ms.""" + + assistant_audio: bytes + """Exactly one tick's worth of PCM16, silence-padded when the assistant was quiet.""" + + assistant_audio_raw_bytes: int + """Real audio bytes received before padding. Zero means the assistant was silent.""" + + wall_clock_ms: int + """Unix ms at the tick's I/O boundary. For latency metrics only, never for ordering.""" + + skip_item_id: str | None = None + """Provider item whose remaining audio must be discarded after a barge-in.""" + + interruption_audio_start_ms: int | None = None + """Played position where the caller cut in, in simulated ms.""" + + provider_stalled: bool = False + """The assistant has produced nothing for so long that the run is not usable. + + Reported rather than raised. A stall is a bad *record*, not a bad *program*: the + conversation ends with a terminal reason the runner treats as a validation failure + and retries, exactly like any other unfinished record, and the partial audio and + event log survive for diagnosis. + """ + + @property + def has_assistant_speech(self) -> bool: + """Whether any real assistant audio arrived this tick.""" + return self.assistant_audio_raw_bytes > 0 + + +def played_audio_ms(*, ticks_released: int) -> int: + """Assistant audio actually released into the conversation, in simulated ms. + + This is deliberately not "bytes received": a realtime provider generates + faster than real time, so the provider's idea of playback position runs ahead + of what the caller has heard. Truncation must use this number. + """ + return ticks_released * TICK_DURATION_MS + + +class TickAudioSplit(NamedTuple): + """Result of splitting audio at a tick boundary.""" + + chunk: bytes + """Always exactly bytes_per_tick, silence-padded when the input was short.""" + + overflow: bytes + """Any remainder past bytes_per_tick, carried to the next tick.""" + + +def split_tick_audio(audio: bytes, bytes_per_tick: int = BYTES_PER_TICK) -> TickAudioSplit: + """Split audio into exactly one tick's worth plus overflow, padding short input with silence.""" + if len(audio) >= bytes_per_tick: + return TickAudioSplit(audio[:bytes_per_tick], audio[bytes_per_tick:]) + return TickAudioSplit(audio + SILENCE_BYTE * (bytes_per_tick - len(audio)), b"") diff --git a/src/eva/user_simulator/cascade/tts.py b/src/eva/user_simulator/cascade/tts.py new file mode 100644 index 00000000..a6c5b496 --- /dev/null +++ b/src/eva/user_simulator/cascade/tts.py @@ -0,0 +1,70 @@ +"""Caller speech synthesis via Cartesia Sonic.""" + +from __future__ import annotations + +import os +from collections.abc import AsyncIterator +from typing import Any + +import httpx + +from eva.user_simulator.cascade.constants import CALLER_SAMPLE_RATE +from eva.utils.logging import get_logger + +logger = get_logger(__name__) + +CARTESIA_URL = "https://api.cartesia.ai/tts/bytes" +CARTESIA_VERSION = "2024-06-10" +DEFAULT_FEMALE_VOICE = "f786b574-daa5-4673-aa0c-cbe3e8534c02" +DEFAULT_MALE_VOICE = "a0e99841-438c-4a64-b679-ae501e7d6091" +_FEMALE_PERSONA_ID = 1 + + +class CartesiaTTS: + """Renders caller text to PCM16 at the simulator's sample rate.""" + + def __init__(self, params: dict[str, Any], *, language: str = "en") -> None: + self._model = params.get("model", "sonic-3.5") + self._api_key = params.get("api_key") or os.environ.get("CARTESIA_API_KEY", "") + self._female_voice = params.get("female_voice", DEFAULT_FEMALE_VOICE) + self._male_voice = params.get("male_voice", DEFAULT_MALE_VOICE) + self._language = language + + def voice_for_persona(self, persona_config: dict[str, Any]) -> str: + """Pick a stable voice for this persona, mirroring the existing gender scheme.""" + if persona_config.get("user_persona_id") == _FEMALE_PERSONA_ID: + return self._female_voice + if persona_config.get("user_persona_id") is None: + return self._female_voice + return self._male_voice + + async def stream(self, text: str, *, voice_id: str) -> AsyncIterator[bytes]: + """Yield PCM16 chunks as they render, so playout can start before the tail exists.""" + if not text: + return + if not self._api_key: + raise ValueError("Cartesia API key missing: set tts_params.api_key or CARTESIA_API_KEY") + + body = { + "model_id": self._model, + "transcript": text, + "voice": {"mode": "id", "id": voice_id}, + "language": self._language, + "output_format": { + "container": "raw", + "encoding": "pcm_s16le", + "sample_rate": CALLER_SAMPLE_RATE, + }, + } + headers = {"X-API-Key": self._api_key, "Cartesia-Version": CARTESIA_VERSION} + + async with httpx.AsyncClient(timeout=30.0) as client: + async with client.stream("POST", CARTESIA_URL, json=body, headers=headers) as response: + response.raise_for_status() + async for chunk in response.aiter_bytes(): + if chunk: + yield chunk + + async def synthesize(self, text: str, *, voice_id: str) -> bytes: + """Render text to raw PCM16 mono at CALLER_SAMPLE_RATE.""" + return b"".join([chunk async for chunk in self.stream(text, voice_id=voice_id)]) diff --git a/src/eva/user_simulator/factory.py b/src/eva/user_simulator/factory.py index d2cef7e2..ca6c366f 100644 --- a/src/eva/user_simulator/factory.py +++ b/src/eva/user_simulator/factory.py @@ -4,7 +4,12 @@ from typing import Any -from eva.models.config import ElevenLabsSimulatorConfig, OpenAIRealtimeSimulatorConfig, UserSimulatorConfig +from eva.models.config import ( + CascadeSimulatorConfig, + ElevenLabsSimulatorConfig, + OpenAIRealtimeSimulatorConfig, + UserSimulatorConfig, +) from eva.user_simulator.base import AbstractUserSimulator @@ -12,7 +17,13 @@ def create_user_simulator( simulator_config: UserSimulatorConfig, **kwargs: Any, ) -> AbstractUserSimulator: - """Create the configured simulated caller without importing unused providers.""" + """Create the configured simulated caller without importing unused providers. + + ``framework`` names the assistant framework and is consumed here rather than + forwarded: only the cascade caller varies its transport by framework, and the + other two providers would raise on the unexpected keyword. + """ + framework = kwargs.pop("framework", "pipecat") if isinstance(simulator_config, ElevenLabsSimulatorConfig): from eva.user_simulator.elevenlabs import ElevenLabsUserSimulator @@ -21,4 +32,8 @@ def create_user_simulator( from eva.user_simulator.openai_realtime import OpenAIRealtimeUserSimulator return OpenAIRealtimeUserSimulator(simulator_config=simulator_config, **kwargs) + if isinstance(simulator_config, CascadeSimulatorConfig): + from eva.user_simulator.cascade.simulator import CascadeUserSimulator + + return CascadeUserSimulator(simulator_config=simulator_config, framework=framework, **kwargs) raise ValueError(f"Unknown user simulator provider: {simulator_config.provider!r}") diff --git a/tests/unit/assistant/test_base_server_pacing.py b/tests/unit/assistant/test_base_server_pacing.py new file mode 100644 index 00000000..54a2e783 --- /dev/null +++ b/tests/unit/assistant/test_base_server_pacing.py @@ -0,0 +1,40 @@ +import inspect + +from eva.assistant.base_server import AbstractAssistantServer + + +def test_paced_output_defaults_to_true(): + # The existing ElevenLabs caller depends on real-time cadence for its + # silence heuristics, so the default must not change. + signature = inspect.signature(AbstractAssistantServer.__init__) + assert signature.parameters["paced_output"].default is True + + +def test_every_server_accepts_paced_output(): + # worker.py always passes paced_output; a server that omits it from its signature + # raises TypeError at construction (this is what broke the pipecat run). + import inspect + + from eva.assistant.elevenlabs_server import ElevenLabsAssistantServer + from eva.assistant.gemini_live_server import GeminiLiveAssistantServer + from eva.assistant.pipecat_server import PipecatAssistantServer + from eva.assistant.smallest_hydra_server import SmallestHydraAssistantServer + + for server in ( + PipecatAssistantServer, + GeminiLiveAssistantServer, + ElevenLabsAssistantServer, + SmallestHydraAssistantServer, + ): + params = inspect.signature(server.__init__).parameters + assert "paced_output" in params, f"{server.__name__} rejects paced_output" + + +def test_only_servers_owning_their_throttle_claim_unpaced_support(): + from eva.assistant.elevenlabs_server import ElevenLabsAssistantServer + from eva.assistant.openai_realtime_server import OpenAIRealtimeAssistantServer + from eva.assistant.pipecat_server import PipecatAssistantServer + + assert OpenAIRealtimeAssistantServer.supports_unpaced_output is True + assert PipecatAssistantServer.supports_unpaced_output is False + assert ElevenLabsAssistantServer.supports_unpaced_output is False diff --git a/tests/unit/assistant/test_openai_realtime_pacing.py b/tests/unit/assistant/test_openai_realtime_pacing.py new file mode 100644 index 00000000..1c29b046 --- /dev/null +++ b/tests/unit/assistant/test_openai_realtime_pacing.py @@ -0,0 +1,143 @@ +import asyncio +import time + +import pytest + +from eva.assistant.openai_realtime_server import OpenAIRealtimeAssistantServer + + +class FakeWS: + def __init__(self) -> None: + self.sent: list[str] = [] + + async def send_text(self, message: str) -> None: + self.sent.append(message) + + +def _server(paced: bool, tmp_path) -> OpenAIRealtimeAssistantServer: + from eva.models.agents import AgentConfig + from eva.models.config import ModelConfig + + db_path = tmp_path / "db.json" + db_path.write_text("{}") + + return OpenAIRealtimeAssistantServer( + current_date_time="2026-01-01T00:00:00", + pipeline_config=ModelConfig(s2s="gpt-realtime", s2s_params={"model": "gpt-realtime", "api_key": "k"}), + agent=AgentConfig( + id="agent_itsm", + name="agent_itsm", + role="r", + description="d", + instructions="i", + tool_module_path="eva.assistant.tools.itsm_tools", + ), + agent_config_path="configs/agents/itsm_agent.yaml", + scenario_db_path=str(db_path), + output_dir=tmp_path, + port=9999, + conversation_id="c1", + paced_output=paced, + ) + + +@pytest.mark.asyncio +async def test_unpaced_output_drains_without_sleeping(tmp_path): + server = _server(paced=False, tmp_path=tmp_path) + server._running = True + ws = FakeWS() + queue: asyncio.Queue[bytes] = asyncio.Queue() + for _ in range(20): + queue.put_nowait(b"\xff" * 160) + + started = time.monotonic() + task = asyncio.create_task(server._pace_audio_output(ws, queue)) + await asyncio.sleep(0.05) + task.cancel() + + # 20 paced chunks would take ~400ms; unpaced must finish inside the 50ms window. + assert len(ws.sent) == 20 + assert time.monotonic() - started < 0.2 + + +@pytest.mark.asyncio +async def test_paced_output_still_holds_real_time_cadence(tmp_path): + server = _server(paced=True, tmp_path=tmp_path) + server._running = True + ws = FakeWS() + queue: asyncio.Queue[bytes] = asyncio.Queue() + for _ in range(20): + queue.put_nowait(b"\xff" * 160) + + task = asyncio.create_task(server._pace_audio_output(ws, queue)) + await asyncio.sleep(0.05) + task.cancel() + + # ~20ms per chunk means only a handful land in a 50ms window. + assert len(ws.sent) < 10 + + +def test_user_recently_active_counts_audio_deltas_not_wall_time(tmp_path): + server = _server(paced=False, tmp_path=tmp_path) + + server.note_user_audio() + assert server.user_recently_active() is True + + # A long stall must not make a just-spoken user look idle. + for _ in range(server.USER_ACTIVE_GUARD_DELTAS): + server.note_assistant_delta() + assert server.user_recently_active() is False + + +class FakeConn: + """Records conversation.item.truncate calls.""" + + def __init__(self) -> None: + self.truncated: list[dict] = [] + outer = self + + class _Item: + async def truncate(self, *, item_id, content_index, audio_end_ms): + outer.truncated.append( + {"item_id": item_id, "content_index": content_index, "audio_end_ms": audio_end_ms} + ) + + class _Conversation: + item = _Item() + + self.conversation = _Conversation() + + +@pytest.mark.asyncio +async def test_truncate_targets_the_item_currently_producing_audio(tmp_path): + server = _server(paced=False, tmp_path=tmp_path) + server._assistant_state.active_item_id = "item_42" + conn = FakeConn() + + await server._truncate_response(conn, 600) + + assert conn.truncated == [{"item_id": "item_42", "content_index": 0, "audio_end_ms": 600}] + + +@pytest.mark.asyncio +async def test_truncate_is_a_no_op_when_no_item_is_active(tmp_path): + server = _server(paced=False, tmp_path=tmp_path) + conn = FakeConn() + + await server._truncate_response(conn, 600) + + assert conn.truncated == [] + + +@pytest.mark.asyncio +async def test_audio_delta_records_the_active_item(tmp_path): + import base64 + from types import SimpleNamespace + + server = _server(paced=False, tmp_path=tmp_path) + queue: asyncio.Queue[bytes] = asyncio.Queue() + event = SimpleNamespace(delta=base64.b64encode(b"\x00" * 480).decode(), item_id="item_7") + + await server._on_audio_delta(event, queue) + + assert server._assistant_state.active_item_id == "item_7" diff --git a/tests/unit/models/test_config_models.py b/tests/unit/models/test_config_models.py index d797dbcd..548fc309 100644 --- a/tests/unit/models/test_config_models.py +++ b/tests/unit/models/test_config_models.py @@ -1225,3 +1225,75 @@ def test_openai_realtime_rejects_accent_perturbation_during_config_load(self): user_simulator={"provider": "openai_realtime"}, perturbation={"accent": "french"}, ) + + def test_cascade_rejects_accent_perturbation_during_config_load(self): + with pytest.raises(ValidationError, match="Accent perturbations require the ElevenLabs user simulator"): + _config( + env_vars=_BASE_ENV, + user_simulator={"provider": "cascade"}, + perturbation={"accent": "french"}, + ) + + def test_cascade_nested_environment_configuration(self): + from eva.models.config import CascadeSimulatorConfig + + config = _config( + env_vars=_BASE_ENV + | { + "EVA_USER_SIMULATOR__PROVIDER": "cascade", + "EVA_USER_SIMULATOR__STT_PARAMS": json.dumps({"model": "x"}), + } + ) + + assert isinstance(config.user_simulator, CascadeSimulatorConfig) + assert config.user_simulator.stt_params == {"model": "x"} + + +def test_cascade_simulator_config_defaults(): + from eva.models.config import CascadeSimulatorConfig + + config = CascadeSimulatorConfig() + + assert config.provider == "cascade" + assert config.stt == "elevenlabs" + assert config.stt_params["model"] == "scribe_v2_realtime" + assert config.tts == "cartesia" + assert config.tts_params["model"] == "sonic-3.5" + assert config.llm == "user-llm" + + +def test_user_simulator_union_discriminates_cascade(): + from pydantic import TypeAdapter + + from eva.models.config import CascadeSimulatorConfig, UserSimulatorConfig + + parsed = TypeAdapter(UserSimulatorConfig).validate_python({"provider": "cascade"}) + + assert isinstance(parsed, CascadeSimulatorConfig) + + +def test_cascade_behaviors_default_off(): + from eva.models.config import CascadeSimulatorConfig + + config = CascadeSimulatorConfig() + + assert config.enable_backchannel is False + assert config.enable_interruptions is False + assert config.speculative_generation is False + + +def test_cascade_behaviors_can_be_enabled_independently(): + from eva.models.config import CascadeSimulatorConfig + + config = CascadeSimulatorConfig(enable_backchannel=True) + + assert config.enable_backchannel is True + assert config.enable_interruptions is False + + +def test_cascade_decision_llm_defaults_to_the_caller_llm(): + from eva.models.config import CascadeSimulatorConfig + + config = CascadeSimulatorConfig() + + assert config.decision_llm == config.llm diff --git a/tests/unit/orchestrator/test_worker.py b/tests/unit/orchestrator/test_worker.py index e5a0366d..f7c2bee3 100644 --- a/tests/unit/orchestrator/test_worker.py +++ b/tests/unit/orchestrator/test_worker.py @@ -252,6 +252,7 @@ async def test_worker_uses_configured_factory_and_timeout(self, tmp_path, monkey timeout=worker._conversation_guard_timeout_seconds(), perturbation_config=None, language="en", + framework=worker.config.framework, ) def test_worker_timeout_reserves_provider_cleanup_window(self, tmp_path): @@ -342,3 +343,25 @@ async def test_stats_captured_on_time_limit_exceeded(self, tmp_path): assert result.num_turns == 3 assert result.num_tool_calls == 1 assert result.conversation_ended_reason == "time_limit_exceeded" + + +def test_pacing_disabled_for_tick_driven_cascade_runs(): + from eva.models.config import CascadeSimulatorConfig + from eva.orchestrator.worker import should_pace_assistant_output + + assert should_pace_assistant_output(CascadeSimulatorConfig(), framework="openai_realtime") is False + + +def test_pacing_kept_for_cascade_on_a_real_time_framework(): + from eva.models.config import CascadeSimulatorConfig + from eva.orchestrator.worker import should_pace_assistant_output + + assert should_pace_assistant_output(CascadeSimulatorConfig(), framework="pipecat") is True + + +def test_pacing_kept_for_the_elevenlabs_simulator(): + from eva.models.config import ElevenLabsSimulatorConfig + from eva.orchestrator.worker import should_pace_assistant_output + + # Its silence heuristics read cadence; unpacing it would break turn detection. + assert should_pace_assistant_output(ElevenLabsSimulatorConfig(), framework="openai_realtime") is True diff --git a/tests/unit/user_simulator/cascade/__init__.py b/tests/unit/user_simulator/cascade/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/unit/user_simulator/cascade/conftest.py b/tests/unit/user_simulator/cascade/conftest.py new file mode 100644 index 00000000..038040e3 --- /dev/null +++ b/tests/unit/user_simulator/cascade/conftest.py @@ -0,0 +1,17 @@ +import pytest + +from eva.user_simulator.cascade.phrase_cache import PhraseCache + + +@pytest.fixture(autouse=True) +def _isolate_phrase_cache(): + """Clear the process-global phrase audio between tests. + + The cache is shared across conversations on purpose — that is what stops every + record re-rendering the same "mm-hmm" — but tests are conversations too, so + without this one test's renders satisfy the next one's prerender and call + counts come out wrong. + """ + PhraseCache.clear() + yield + PhraseCache.clear() diff --git a/tests/unit/user_simulator/cascade/test_adapter_base.py b/tests/unit/user_simulator/cascade/test_adapter_base.py new file mode 100644 index 00000000..bac37908 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_adapter_base.py @@ -0,0 +1,36 @@ +import pytest + +from eva.user_simulator.cascade.adapter.base import Adapter + + +def test_adapter_cannot_be_instantiated_directly(): + with pytest.raises(TypeError): + Adapter() + + +async def test_concrete_adapter_satisfies_the_interface(): + from eva.user_simulator.cascade.tick_result import TickResult + + class StubAdapter(Adapter): + async def start(self) -> None: + pass + + async def run_tick( + self, tick_number: int, outgoing_audio: bytes | None, *, barge_in: bool = False + ) -> TickResult: + return TickResult( + tick_number=tick_number, + assistant_audio=b"\x00" * 4, + assistant_audio_raw_bytes=0, + wall_clock_ms=0, + ) + + async def stop(self) -> None: + pass + + adapter = StubAdapter() + await adapter.start() + result = await adapter.run_tick(0, None) + await adapter.stop() + + assert result.tick_number == 0 diff --git a/tests/unit/user_simulator/cascade/test_constants.py b/tests/unit/user_simulator/cascade/test_constants.py new file mode 100644 index 00000000..1aeacec0 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_constants.py @@ -0,0 +1,43 @@ +from eva.user_simulator.cascade.constants import ( + BYTES_PER_TICK, + TICK_DURATION_MS, + WAIT_TO_RESPOND_OTHER_MS, + WAIT_TO_RESPOND_SELF_MS, + ms_to_ticks, +) + +THRESHOLD_MS_CONSTANTS = [ + WAIT_TO_RESPOND_OTHER_MS, + WAIT_TO_RESPOND_SELF_MS, +] + + +def test_bytes_per_tick_is_one_tick_of_pcm16_at_16khz(): + # 16000 samples/s * 0.2s * 2 bytes/sample + assert BYTES_PER_TICK == 6400 + + +def test_threshold_constants_are_exact_multiples_of_tick_duration(): + for threshold_ms in THRESHOLD_MS_CONSTANTS: + assert threshold_ms % TICK_DURATION_MS == 0 + + +def test_ms_to_ticks_converts_and_floors_sub_tick_remainder(): + assert ms_to_ticks(WAIT_TO_RESPOND_OTHER_MS) == 5 + assert ms_to_ticks(150) == 0 + + +def test_listener_check_interval_is_two_seconds_in_ticks(): + from eva.user_simulator.cascade.constants import LISTENER_CHECK_INTERVAL_MS, ms_to_ticks + + assert LISTENER_CHECK_INTERVAL_MS == 2000 + assert ms_to_ticks(LISTENER_CHECK_INTERVAL_MS) == 10 + + +def test_the_vocabularies_are_no_longer_constants(): + # They moved to configs/caller_phrases.yaml because they are language data, not + # timing. Timing constants staying here is the whole distinction. + from eva.user_simulator.cascade import constants + + assert not hasattr(constants, "BACKCHANNEL_PHRASES") + assert not hasattr(constants, "BARGE_IN_OPENERS") diff --git a/tests/unit/user_simulator/cascade/test_decision_log.py b/tests/unit/user_simulator/cascade/test_decision_log.py new file mode 100644 index 00000000..7ca3311b --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_decision_log.py @@ -0,0 +1,78 @@ +"""Tests for the caller's out-of-turn decision trace.""" + +import json + +from eva.user_simulator.cascade.decision_log import DecisionLog + + +def test_rows_are_written_one_json_object_per_line(tmp_path): + log = DecisionLog(tmp_path / "trace.jsonl") + log.log("tick", tick=1, has_assistant_speech=True) + log.log("listener_check", tick=10, should_interrupt=False) + + log.save() + + rows = [json.loads(line) for line in (tmp_path / "trace.jsonl").read_text().splitlines()] + assert [r["kind"] for r in rows] == ["tick", "listener_check"] + assert rows[0]["has_assistant_speech"] is True + + +def test_no_file_is_written_when_nothing_was_traced(tmp_path): + DecisionLog(tmp_path / "trace.jsonl").save() + + assert not (tmp_path / "trace.jsonl").exists() + + +def test_rows_are_readable_before_save_so_a_crashed_run_still_has_a_trace(tmp_path): + log = DecisionLog(tmp_path / "trace.jsonl") + log.log("tick", tick=1) + + # No save() call: this is the killed-mid-run case. + assert json.loads((tmp_path / "trace.jsonl").read_text().splitlines()[0])["tick"] == 1 + + +def test_save_is_idempotent(tmp_path): + log = DecisionLog(tmp_path / "trace.jsonl") + log.log("tick", tick=1) + + log.save() + log.save() + + +def test_summary_counts_rows_by_kind(tmp_path): + log = DecisionLog(tmp_path / "trace.jsonl") + for _ in range(3): + log.log("tick") + log.log("listener_check") + + assert log.summary() == {"tick": 3, "listener_check": 1} + + +async def test_a_declined_check_is_still_recorded_with_its_raw_reply(): + # The whole point: a NO must be distinguishable from a check that never ran. + from eva.user_simulator.cascade.decisions import ListenerDecisions + from tests.unit.user_simulator.cascade.test_decisions import FakeLLM + + decisions = ListenerDecisions( + FakeLLM(["NO", "NO"]), interrupt_prompt="{conversation_history}", backchannel_prompt="{conversation_history}" + ) + + verdict = await decisions.evaluate("agent talking", allow_interrupt=True, allow_backchannel=True) + + assert verdict.should_interrupt is False + assert verdict.interrupt_trace.ran is True + assert verdict.interrupt_trace.raw == "NO" + + +async def test_a_check_that_was_not_allowed_reports_that_it_never_ran(): + from eva.user_simulator.cascade.decisions import ListenerDecisions + from tests.unit.user_simulator.cascade.test_decisions import FakeLLM + + decisions = ListenerDecisions( + FakeLLM([]), interrupt_prompt="{conversation_history}", backchannel_prompt="{conversation_history}" + ) + + verdict = await decisions.evaluate("agent talking", allow_interrupt=False, allow_backchannel=False) + + assert verdict.interrupt_trace.ran is False + assert verdict.interrupt_trace.raw == "" diff --git a/tests/unit/user_simulator/cascade/test_decisions.py b/tests/unit/user_simulator/cascade/test_decisions.py new file mode 100644 index 00000000..7145c06a --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_decisions.py @@ -0,0 +1,155 @@ +from eva.user_simulator.cascade.decisions import ListenerDecisions, parse_yes_no + + +def test_parse_yes_no_accepts_plain_yes(): + assert parse_yes_no("YES") is True + + +def test_parse_yes_no_is_case_and_whitespace_insensitive(): + assert parse_yes_no(" yes\n") is True + + +def test_parse_yes_no_treats_anything_else_as_no(): + assert parse_yes_no("NO") is False + assert parse_yes_no("maybe") is False + assert parse_yes_no("") is False + + +class FakeLLM: + """Returns scripted replies, or raises when configured to.""" + + def __init__(self, replies: list[str] | None = None, error: Exception | None = None) -> None: + self.replies = replies or [] + self.error = error + self.calls = 0 + + async def decide(self, prompt: str) -> str: + self.calls += 1 + if self.error is not None: + raise self.error + return self.replies.pop(0) if self.replies else "NO" + + +async def test_both_checks_run_and_interrupt_wins_ties(): + llm = FakeLLM(["YES", "YES"]) + decisions = ListenerDecisions( + llm, interrupt_prompt="i {conversation_history}", backchannel_prompt="b {conversation_history}" + ) + + verdict = await decisions.evaluate("AGENT: hello", allow_interrupt=True, allow_backchannel=True) + + assert verdict.should_interrupt is True + assert verdict.should_backchannel is False + assert llm.calls == 2 + + +async def test_backchannel_alone_when_interrupt_declines(): + llm = FakeLLM(["NO", "YES"]) + decisions = ListenerDecisions( + llm, interrupt_prompt="i {conversation_history}", backchannel_prompt="b {conversation_history}" + ) + + verdict = await decisions.evaluate("AGENT: hello", allow_interrupt=True, allow_backchannel=True) + + assert verdict.should_interrupt is False + assert verdict.should_backchannel is True + + +async def test_disabled_behaviors_are_not_called_at_all(): + llm = FakeLLM(["YES"]) + decisions = ListenerDecisions( + llm, interrupt_prompt="i {conversation_history}", backchannel_prompt="b {conversation_history}" + ) + + verdict = await decisions.evaluate("AGENT: hello", allow_interrupt=False, allow_backchannel=True) + + assert verdict.should_interrupt is False + assert llm.calls == 1 + + +async def test_a_failing_check_fails_closed(): + llm = FakeLLM(error=RuntimeError("provider down")) + decisions = ListenerDecisions( + llm, interrupt_prompt="i {conversation_history}", backchannel_prompt="b {conversation_history}" + ) + + verdict = await decisions.evaluate("AGENT: hello", allow_interrupt=True, allow_backchannel=True) + + assert verdict.should_interrupt is False + assert verdict.should_backchannel is False + + +async def test_the_goal_is_substituted_into_the_interrupt_prompt(): + llm = FakeLLM(["NO", "NO"]) + seen: list[str] = [] + + class _Recorder(FakeLLM): + async def decide(self, prompt: str) -> str: + seen.append(prompt) + return await super().decide(prompt) + + decisions = ListenerDecisions( + _Recorder(["NO", "NO"]), + interrupt_prompt="goal={user_goal} history={conversation_history}", + backchannel_prompt="b {conversation_history}", + user_goal="Unlock my account.", + ) + + await decisions.evaluate("AGENT: hello", allow_interrupt=True, allow_backchannel=True) + + assert "goal=Unlock my account." in seen[0] + assert llm.calls == 0 + + +async def test_a_backchannel_prompt_without_a_goal_slot_still_works(): + # str.format ignores unused keyword arguments, so one signature serves both prompts. + decisions = ListenerDecisions( + FakeLLM(["NO", "YES"]), + interrupt_prompt="i {conversation_history}", + backchannel_prompt="b {conversation_history}", + user_goal="Unlock my account.", + ) + + verdict = await decisions.evaluate("AGENT: hello", allow_interrupt=True, allow_backchannel=True) + + assert verdict.should_backchannel is True + + +def test_goal_summary_is_compact_and_names_what_is_left(): + from eva.user_simulator.cascade.simulator import summarize_goal + + summary = summarize_goal( + { + "high_level_user_goal": "Unlock my AD account.", + "decision_tree": { + "must_have_criteria": ["account unlocked"], + "nice_to_have_criteria": ["a case number"], + "negotiation_behavior": "take the fastest fix offered", + "resolution_condition": "user can sign in", + "failure_condition": "agent cannot unlock it", + "escalation_behavior": "do not ask for a live agent", + "edge_cases": ["a very long irrelevant edge case " * 40], + }, + } + ) + + for expected in ( + "Unlock my AD account.", + "account unlocked", + "a case number", + "take the fastest fix offered", + "user can sign in", + "agent cannot unlock it", + "do not ask for a live agent", + ): + assert expected in summary, expected + # edge_cases is long and describes how to answer questions, not whether the goal is done. + assert "irrelevant" not in summary + + +def test_goal_summary_omits_absent_fields_without_blank_labels(): + from eva.user_simulator.cascade.simulator import summarize_goal + + summary = summarize_goal({"high_level_user_goal": "Unlock my account.", "decision_tree": {}}) + + assert summary == "GOAL: Unlock my account." diff --git a/tests/unit/user_simulator/cascade/test_phrase_cache.py b/tests/unit/user_simulator/cascade/test_phrase_cache.py new file mode 100644 index 00000000..b951976e --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_phrase_cache.py @@ -0,0 +1,101 @@ +import pytest + +from eva.user_simulator.cascade.phrase_cache import PhraseCache + + +class FakeTTS: + """Counts synthesis calls and returns deterministic audio per phrase.""" + + def __init__(self) -> None: + self.calls: list[str] = [] + + async def synthesize(self, text: str, *, voice_id: str) -> bytes: + self.calls.append(text) + return text.encode() + + +async def test_prerender_synthesizes_every_phrase_once(): + tts = FakeTTS() + cache = PhraseCache(tts, voice_id="voice-f") + + await cache.prerender(["uh-huh", "mm-hmm"]) + + assert sorted(tts.calls) == ["mm-hmm", "uh-huh"] + + +async def test_cached_audio_is_returned_without_further_synthesis(): + tts = FakeTTS() + cache = PhraseCache(tts, voice_id="voice-f") + await cache.prerender(["uh-huh"]) + + audio = cache.get("uh-huh") + + assert audio == b"uh-huh" + assert len(tts.calls) == 1 + + +async def test_choose_returns_a_phrase_from_the_cache_deterministically(): + tts = FakeTTS() + cache = PhraseCache(tts, voice_id="voice-f", seed=7) + await cache.prerender(["uh-huh", "mm-hmm"]) + + first = cache.choose(["uh-huh", "mm-hmm"]) + replay = PhraseCache(FakeTTS(), voice_id="voice-f", seed=7) + await replay.prerender(["uh-huh", "mm-hmm"]) + + assert first == replay.choose(["uh-huh", "mm-hmm"]) + + +async def test_requesting_an_unrendered_phrase_raises(): + cache = PhraseCache(FakeTTS(), voice_id="voice-f") + + with pytest.raises(KeyError, match="not pre-rendered"): + cache.get("never-rendered") + + +async def test_a_second_conversation_on_the_same_voice_renders_nothing(): + # This is the point of the global cache: a run of N records against one voice + # renders each phrase once, not N times. + tts = FakeTTS() + first = PhraseCache(tts, voice_id="voice-f") + await first.prerender(["uh-huh", "mm-hmm"]) + + second = PhraseCache(tts, voice_id="voice-f") + await second.prerender(["uh-huh", "mm-hmm"]) + + assert len(tts.calls) == 2 + assert second.get("uh-huh") == b"uh-huh" + + +async def test_only_the_phrases_a_voice_is_missing_are_rendered(): + tts = FakeTTS() + await PhraseCache(tts, voice_id="voice-f").prerender(["uh-huh"]) + tts.calls.clear() + + await PhraseCache(tts, voice_id="voice-f").prerender(["uh-huh", "mm-hmm"]) + + assert tts.calls == ["mm-hmm"] + + +async def test_each_voice_keeps_its_own_audio(): + # Two genders means two voices; one must never be served the other's audio. + tts = FakeTTS() + female = PhraseCache(tts, voice_id="voice-f") + male = PhraseCache(tts, voice_id="voice-m") + await female.prerender(["uh-huh"]) + await male.prerender(["uh-huh"]) + + assert len(tts.calls) == 2 + assert female.get("uh-huh") == b"uh-huh" + assert male.get("uh-huh") == b"uh-huh" + + +async def test_concurrent_conversations_do_not_double_render(): + import asyncio + + tts = FakeTTS() + caches = [PhraseCache(tts, voice_id="voice-f") for _ in range(4)] + + await asyncio.gather(*(c.prerender(["uh-huh", "mm-hmm"]) for c in caches)) + + assert sorted(tts.calls) == ["mm-hmm", "uh-huh"] diff --git a/tests/unit/user_simulator/cascade/test_phrases.py b/tests/unit/user_simulator/cascade/test_phrases.py new file mode 100644 index 00000000..60ca00f3 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_phrases.py @@ -0,0 +1,88 @@ +import pytest +import yaml + +from eva.user_simulator.cascade import phrases as phrases_module +from eva.user_simulator.cascade.phrases import ( + PHRASES_PATH, + CallerPhrases, + candidate_languages, + load_phrases, +) + + +@pytest.fixture(autouse=True) +def _clear_phrase_cache(): + phrases_module._cache.clear() + yield + phrases_module._cache.clear() + + +@pytest.fixture +def phrase_file(tmp_path, monkeypatch): + """Point the loader at a temporary phrase file.""" + + def write(data): + path = tmp_path / "caller_phrases.yaml" + path.write_text(yaml.safe_dump(data, allow_unicode=True), encoding="utf-8") + monkeypatch.setattr(phrases_module, "PHRASES_PATH", path) + return path + + return write + + +def test_the_shipped_file_has_english(): + # English is the fallback for every language, so it is the one entry that must exist. + data = yaml.safe_load(PHRASES_PATH.read_text(encoding="utf-8")) + + assert data["en"]["backchannels"] + assert data["en"]["barge_in_openers"] + + +def test_a_language_with_its_own_entry_uses_it(phrase_file): + phrase_file( + { + "en": {"backchannels": ["uh-huh"], "barge_in_openers": ["Wait—"]}, + "fr": {"backchannels": ["hmm", "ouais"], "barge_in_openers": ["Attendez—"]}, + } + ) + + assert load_phrases("fr") == CallerPhrases(backchannels=["hmm", "ouais"], barge_in_openers=["Attendez—"]) + + +def test_a_regional_variant_falls_back_to_its_base_language(phrase_file): + # fr-CA shares its continuers with fr; falling all the way to English would be worse. + phrase_file( + { + "en": {"backchannels": ["uh-huh"], "barge_in_openers": ["Wait—"]}, + "fr": {"backchannels": ["hmm"], "barge_in_openers": ["Attendez—"]}, + } + ) + + assert load_phrases("fr-CA").backchannels == ["hmm"] + + +def test_an_unknown_language_falls_back_to_english_rather_than_failing(phrase_file): + # A missing translation should degrade the behavior's realism, not abort the run. + phrase_file({"en": {"backchannels": ["uh-huh"], "barge_in_openers": ["Wait—"]}}) + + assert load_phrases("ja").backchannels == ["uh-huh"] + + +def test_a_missing_file_yields_an_empty_vocabulary(phrase_file, tmp_path, monkeypatch): + monkeypatch.setattr(phrases_module, "PHRASES_PATH", tmp_path / "absent.yaml") + + result = load_phrases("en") + + assert result.vocabulary == [] + + +def test_candidate_order_is_most_specific_first(): + assert candidate_languages("fr-CA") == ["fr-CA", "fr", "en"] + assert candidate_languages("fr") == ["fr", "en"] + assert candidate_languages("en") == ["en"] + + +def test_vocabulary_is_everything_that_needs_rendering(phrase_file): + phrase_file({"en": {"backchannels": ["uh-huh", "mm-hmm"], "barge_in_openers": ["Wait—"]}}) + + assert load_phrases("en").vocabulary == ["uh-huh", "mm-hmm", "Wait—"] diff --git a/tests/unit/user_simulator/cascade/test_prompt.py b/tests/unit/user_simulator/cascade/test_prompt.py new file mode 100644 index 00000000..f3ad979b --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_prompt.py @@ -0,0 +1,53 @@ +from eva.utils.prompt_manager import PromptManager + + +def test_cascade_reuses_the_shared_end_call_description(): + # A cascade-specific copy would drift from the other providers' hang-up rules. + from eva.user_simulator.cascade.simulator import END_CALL_DESCRIPTION as cascade_description + from eva.user_simulator.openai_realtime import END_CALL_DESCRIPTION as shared_description + + assert cascade_description is shared_description + + +def test_the_turn_call_carries_no_cascade_specific_contract(): + # The per-domain user_simulator prompt already carries persona, goal and end_call rules; + # layering a cascade-only contract on top is what suppressed the end_call tool call. + # Out-of-turn behavior prompts exist, but they are only ever used in their own + # standalone calls — never appended to the system prompt of the turn call. + from eva.user_simulator.cascade.simulator import CascadeUserSimulator + + sim = object.__new__(CascadeUserSimulator) + sim._build_prompt = lambda: "SYSTEM PROMPT" + sim._history = [] + + assert sim._messages()[0]["content"] == "SYSTEM PROMPT" + + +def test_interruption_decision_prompt_has_a_history_slot_and_binary_contract(): + prompt = PromptManager().get_prompt( + "user_simulator.interruption_decision", conversation_history="AGENT: hello", user_goal="Unlock my account." + ) + + assert "AGENT: hello" in prompt + assert "YES" in prompt + assert "NO" in prompt + + +def test_backchannel_decision_prompt_has_a_history_slot_and_frequency_guidance(): + prompt = PromptManager().get_prompt("user_simulator.backchannel_decision", conversation_history="AGENT: hello") + + assert "AGENT: hello" in prompt + assert "CURRENTLY SPEAKING, INCOMPLETE" in prompt + assert "When in doubt, say NO" in prompt + + +def test_interruption_decision_prompt_carries_the_user_goal(): + prompt = PromptManager().get_prompt( + "user_simulator.interruption_decision", + conversation_history="AGENT: hello", + user_goal="Get my account unlocked.", + ) + + assert "Get my account unlocked." in prompt + # The goodbye case the caller kept barging in on. + assert "likely to hang up" in prompt diff --git a/tests/unit/user_simulator/cascade/test_realtime_ws_adapter.py b/tests/unit/user_simulator/cascade/test_realtime_ws_adapter.py new file mode 100644 index 00000000..157fa165 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_realtime_ws_adapter.py @@ -0,0 +1,364 @@ +import asyncio +import base64 +import json + +import pytest + +try: + import audioop +except ImportError: # pragma: no cover - Python 3.13+ + import audioop_lts as audioop + +from eva.user_simulator.cascade.adapter.realtime_ws import FRAMES_PER_TICK, RealtimeWSAdapter + +BYTES_PER_TICK = 6400 +_SETTLE_ROUNDS = 300 + + +class FakeWebSocket: + """Collects sent frames and replays queued inbound frames. + + recv() completes without ever suspending when a frame is already queued — + unlike real `websockets.recv()`. Use SuspendingFakeWebSocket to model that. + """ + + def __init__(self) -> None: + self.sent: list[str] = [] + self.inbound: asyncio.Queue[str] = asyncio.Queue() + self.closed = False + + async def send(self, message: str) -> None: + self.sent.append(message) + + async def recv(self) -> str: + return await self.inbound.get() + + async def close(self) -> None: + self.closed = True + + +class SuspendingFakeWebSocket(FakeWebSocket): + """A recv() that always suspends at least once before returning, like the real thing.""" + + async def recv(self) -> str: + await asyncio.sleep(0) + return await self.inbound.get() + + +class RaisingFakeWebSocket(FakeWebSocket): + """A recv() that raises once a configured number of successful frames have been read.""" + + def __init__(self, fail_after: int) -> None: + super().__init__() + self._fail_after = fail_after + self._count = 0 + + async def recv(self) -> str: + await asyncio.sleep(0) + if self._count >= self._fail_after: + raise ConnectionError("simulated disconnect") + self._count += 1 + return await self.inbound.get() + + +def _media_frame(mulaw: bytes) -> str: + payload = base64.b64encode(mulaw).decode() + return json.dumps({"event": "media", "media": {"payload": payload}}) + + +async def _settle() -> None: + """Give a background receive task many event-loop turns to drain queued frames.""" + for _ in range(_SETTLE_ROUNDS): + await asyncio.sleep(0) + + +async def test_tick_with_no_inbound_audio_yields_padded_silence(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + result = await adapter.run_tick(0, None) + + assert result.assistant_audio_raw_bytes == 0 + assert result.has_assistant_speech is False + assert len(result.assistant_audio) == BYTES_PER_TICK + + await adapter.stop() + + +async def test_receive_loop_drains_frames_from_a_suspending_websocket(): + """Regression test: a suspending recv() must still be drained by the background loop.""" + ws = SuspendingFakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await ws.inbound.put(_media_frame(b"\xff" * 160)) + await adapter.start() + await _settle() + + result = await adapter.run_tick(0, None) + + assert result.has_assistant_speech is True + assert result.assistant_audio_raw_bytes > 0 + + await adapter.stop() + + +async def test_burst_of_frames_ingested_then_released_one_tick_at_a_time(): + ws = SuspendingFakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + frame_count = 20 # 400ms of wire audio delivered as one burst (two ticks' worth). + for _ in range(frame_count): + await ws.inbound.put(_media_frame(b"\xff" * 160)) + await adapter.start() + await _settle() + + first = await adapter.run_tick(0, None) + second = await adapter.run_tick(1, None) + third = await adapter.run_tick(2, None) + + assert len(first.assistant_audio) == BYTES_PER_TICK + assert len(second.assistant_audio) == BYTES_PER_TICK + assert first.has_assistant_speech is True + assert second.has_assistant_speech is True + total_raw = first.assistant_audio_raw_bytes + second.assistant_audio_raw_bytes + third.assistant_audio_raw_bytes + assert abs(total_raw - frame_count * 640) <= 2 + + await adapter.stop() + + +async def test_receive_error_surfaces_from_run_tick(): + ws = RaisingFakeWebSocket(fail_after=0) + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + await _settle() + + with pytest.raises(RuntimeError): + await adapter.run_tick(0, None) + + await adapter.stop() + + +async def test_inbound_mulaw_is_converted_and_reported_as_speech(): + ws = SuspendingFakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + # 160 mulaw bytes @8kHz == 20ms, resampled to roughly 640 PCM16 bytes @16kHz. + await ws.inbound.put(_media_frame(b"\xff" * 160)) + await adapter.start() + await _settle() + + result = await adapter.run_tick(0, None) + + assert result.assistant_audio_raw_bytes > 0 + assert result.has_assistant_speech is True + assert len(result.assistant_audio) == BYTES_PER_TICK + + await adapter.stop() + + +async def test_outgoing_caller_audio_is_sent_as_twilio_media_frames(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + await adapter.run_tick(0, b"\x00" * BYTES_PER_TICK) + + media_frames = [json.loads(m) for m in ws.sent if json.loads(m).get("event") == "media"] + # One tick (200ms) is ten 20ms wire frames. + assert len(media_frames) == 10 + + await adapter.stop() + + +async def test_silent_tick_emits_a_full_tick_of_silence_frames(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + await adapter.run_tick(0, None) + + media_frames = [json.loads(m) for m in ws.sent if json.loads(m).get("event") == "media"] + assert len(media_frames) == 10 + for frame in media_frames: + mulaw = base64.b64decode(frame["media"]["payload"]) + pcm = audioop.ulaw2lin(mulaw, 2) + assert audioop.max(pcm, 2) == 0 + + await adapter.stop() + + +async def test_mixed_speak_silent_sequence_has_no_gaps_in_outbound_stream(): + """Regression test: a stall between speech ticks must still send silence, not nothing.""" + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + outgoing_sequence = [b"\x00" * BYTES_PER_TICK, None, None, b"\x00" * BYTES_PER_TICK] + for tick_number, outgoing in enumerate(outgoing_sequence): + await adapter.run_tick(tick_number, outgoing) + + media_frames = [json.loads(m) for m in ws.sent if json.loads(m).get("event") == "media"] + assert len(media_frames) == len(outgoing_sequence) * 10 + + await adapter.stop() + + +async def test_overflow_audio_carries_into_the_next_tick(): + ws = SuspendingFakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + # 400ms of assistant audio arrives at once: 3200 mulaw bytes -> ~12800 PCM bytes, + # more than one tick's worth, so it must span both ticks with real audio in each. + await ws.inbound.put(_media_frame(b"\xff" * 3200)) + await adapter.start() + await _settle() + + first = await adapter.run_tick(0, None) + second = await adapter.run_tick(1, None) + + assert len(first.assistant_audio) == BYTES_PER_TICK + assert len(second.assistant_audio) == BYTES_PER_TICK + assert first.assistant_audio_raw_bytes == BYTES_PER_TICK + assert second.assistant_audio_raw_bytes > 0 + assert first.has_assistant_speech is True + assert second.has_assistant_speech is True + + await adapter.stop() + + +async def test_per_frame_resampling_does_not_accumulate_drift(): + ws = SuspendingFakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + frame_count = 50 + for _ in range(frame_count): + await ws.inbound.put(_media_frame(b"\xff" * 160)) + await adapter.start() + await _settle() + + ideal_bytes = frame_count * 640 + ticks_needed = ideal_bytes // BYTES_PER_TICK + 2 + total_raw_bytes = 0 + for tick_number in range(ticks_needed): + result = await adapter.run_tick(tick_number, None) + total_raw_bytes += result.assistant_audio_raw_bytes + + # 1 sample (2 bytes) of PCM16 warm-up loss is expected for the whole + # stream; per-frame loss must not accumulate beyond that. + assert abs(total_raw_bytes - ideal_bytes) <= 2 + + await adapter.stop() + + +async def test_user_speech_start_emitted_once_on_silence_to_audio_transition(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + await adapter.run_tick(0, None) + await adapter.run_tick(1, b"\x00" * BYTES_PER_TICK) + await adapter.run_tick(2, b"\x00" * BYTES_PER_TICK) + + events = [json.loads(m) for m in ws.sent] + starts = [e for e in events if e.get("event") == "user_speech_start"] + assert len(starts) == 1 + assert isinstance(starts[0]["timestamp_ms"], str) + + await adapter.stop() + + +async def test_silent_tick_paces_to_approximately_one_tick_duration(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + start = asyncio.get_event_loop().time() + await adapter.run_tick(0, None) + elapsed = asyncio.get_event_loop().time() - start + + assert 0.15 < elapsed < 0.4 + + await adapter.stop() + + +async def test_speaking_tick_paces_to_approximately_one_tick_duration_not_double(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + start = asyncio.get_event_loop().time() + await adapter.run_tick(0, b"\x00" * BYTES_PER_TICK) + elapsed = asyncio.get_event_loop().time() - start + + assert 0.15 < elapsed < 0.4 + + await adapter.stop() + + +async def test_user_speech_stop_emitted_once_on_audio_to_silence_transition(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + await adapter.run_tick(0, b"\x00" * BYTES_PER_TICK) + await adapter.run_tick(1, None) + await adapter.run_tick(2, None) + + events = [json.loads(m) for m in ws.sent] + stops = [e for e in events if e.get("event") == "user_speech_stop"] + assert len(stops) == 1 + assert isinstance(stops[0]["timestamp_ms"], str) + + await adapter.stop() + + +class StubPerturbator: + """Stands in for AudioPerturbator with a constant, recognizable noise floor.""" + + has_ambient_noise = True + + def get_ambient_chunk(self, size: int) -> bytes: + return b"\x11" * size + + def apply(self, audio: bytes) -> bytes: + return b"\x22" * len(audio) + + +def _media_payloads(ws) -> list[bytes]: + import base64 + + return [ + base64.b64decode(json.loads(m)["media"]["payload"]) for m in ws.sent if json.loads(m).get("event") == "media" + ] + + +async def test_ambient_noise_replaces_silence_when_the_caller_is_not_speaking(): + ws = FakeWebSocket() + adapter = RealtimeWSAdapter( + websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK, perturbator=StubPerturbator() + ) + + await adapter.run_tick(0, None) + + payloads = _media_payloads(ws) + assert len(payloads) == FRAMES_PER_TICK + # Silence would encode to a constant mulaw byte; ambient noise must not. + assert set(b"".join(payloads)) != {0xFF} + + +async def test_ambient_noise_is_mixed_into_caller_speech(): + ws = FakeWebSocket() + perturbator = StubPerturbator() + adapter = RealtimeWSAdapter( + websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK, perturbator=perturbator + ) + + await adapter.run_tick(0, b"\x01" * BYTES_PER_TICK) + + assert len(_media_payloads(ws)) == FRAMES_PER_TICK + + +async def test_silence_is_still_sent_every_tick_without_a_perturbator(): + # Plan 1: a tick that sends no frames at all makes the assistant's turn detection misfire. + ws = FakeWebSocket() + adapter = RealtimeWSAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + + await adapter.run_tick(0, None) + + assert len(_media_payloads(ws)) == FRAMES_PER_TICK diff --git a/tests/unit/user_simulator/cascade/test_scheduler.py b/tests/unit/user_simulator/cascade/test_scheduler.py new file mode 100644 index 00000000..b59d509b --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_scheduler.py @@ -0,0 +1,327 @@ +from eva.user_simulator.cascade.adapter.base import Adapter +from eva.user_simulator.cascade.scheduler import TickScheduler +from eva.user_simulator.cascade.tick_result import TickResult + +BYTES_PER_TICK = 8 + + +class FakeAdapter(Adapter): + """Replays a scripted sequence of assistant speech/silence and records what it was sent.""" + + def __init__(self, speech_ticks: list[bool]) -> None: + self.speech_ticks = speech_ticks + self.sent: list[bytes | None] = [] + self.received_ticks: list[int] = [] + self.barge_in_ticks: list[int] = [] + + async def start(self) -> None: + pass + + async def run_tick(self, tick_number: int, outgoing_audio: bytes | None, *, barge_in: bool = False) -> TickResult: + self.sent.append(outgoing_audio) + self.received_ticks.append(tick_number) + if barge_in: + self.barge_in_ticks.append(tick_number) + speaking = self.speech_ticks[tick_number] if tick_number < len(self.speech_ticks) else False + return TickResult( + tick_number=tick_number, + assistant_audio=(b"\x01" if speaking else b"\x00") * BYTES_PER_TICK, + assistant_audio_raw_bytes=BYTES_PER_TICK if speaking else 0, + wall_clock_ms=tick_number, + ) + + async def stop(self) -> None: + pass + + +def _scheduler(speech_ticks: list[bool]) -> TickScheduler: + return TickScheduler(FakeAdapter(speech_ticks), bytes_per_tick=BYTES_PER_TICK) + + +async def test_caller_may_open_the_conversation_once_the_assistant_falls_silent(): + # Assistant greets for 3 ticks, then goes quiet. + scheduler = _scheduler([True, True, True]) + + for _ in range(3): + await scheduler.run_tick() + assert scheduler.may_take_turn() is False + + # WAIT_TO_RESPOND_OTHER_MS is 5 ticks and the comparison is strict. + for _ in range(5): + await scheduler.run_tick() + assert scheduler.may_take_turn() is False + + await scheduler.run_tick() + assert scheduler.may_take_turn() is True + + +async def test_assistant_speech_resets_the_silence_counter(): + scheduler = _scheduler([False] * 10 + [True]) + + for _ in range(11): + await scheduler.run_tick() + + assert scheduler.may_take_turn() is False + + +async def test_caller_must_also_wait_out_its_own_silence_threshold(): + scheduler = _scheduler([True, False, True] + [False] * 30) # greet, caller turn, quick reply, then silence + await scheduler.run_tick() # tick 0: assistant greets + scheduler.enqueue_utterance(b"\x02" * BYTES_PER_TICK) + + await scheduler.run_tick() # tick 1: caller speaks this tick + assert scheduler.may_take_turn() is False + + await scheduler.run_tick() # tick 2: assistant replies, clearing the awaiting-reply gate + assert scheduler.may_take_turn() is False + + # Assistant silence is satisfied almost immediately, self-silence needs 25 ticks. + for _ in range(24): + await scheduler.run_tick() + assert scheduler.may_take_turn() is False + + await scheduler.run_tick() + assert scheduler.may_take_turn() is True + + +async def test_caller_does_not_speak_before_the_assistant_has_greeted(): + # The assistant sends the opening message; the caller must not race it. + scheduler = _scheduler([]) + + for _ in range(50): + await scheduler.run_tick() + + assert scheduler.may_take_turn() is False + + +async def test_caller_cannot_take_a_second_turn_while_awaiting_a_reply(): + scheduler = _scheduler([True]) # assistant greets on tick 0, then never speaks again + await scheduler.run_tick() + scheduler.enqueue_utterance(b"\x02" * BYTES_PER_TICK) + await scheduler.run_tick() # caller takes its turn + + for _ in range(100): + await scheduler.run_tick() + assert scheduler.may_take_turn() is False + + +async def test_caller_may_take_a_second_turn_once_the_assistant_replies(): + scheduler = _scheduler([True, False, True] + [False] * 30) + await scheduler.run_tick() # tick 0: assistant greets + scheduler.enqueue_utterance(b"\x02" * BYTES_PER_TICK) + await scheduler.run_tick() # tick 1: caller takes its turn + await scheduler.run_tick() # tick 2: assistant replies + + for _ in range(25): + await scheduler.run_tick() + + assert scheduler.may_take_turn() is True + + +async def test_queued_utterance_drains_one_tick_at_a_time(): + adapter = FakeAdapter([]) + scheduler = TickScheduler(adapter, bytes_per_tick=BYTES_PER_TICK) + scheduler.enqueue_utterance(b"\x02" * (BYTES_PER_TICK * 3)) + + for _ in range(4): + await scheduler.run_tick() + + assert adapter.sent[0] == b"\x02" * BYTES_PER_TICK + assert adapter.sent[1] == b"\x02" * BYTES_PER_TICK + assert adapter.sent[2] == b"\x02" * BYTES_PER_TICK + assert adapter.sent[3] is None + assert scheduler.caller_is_speaking is False + + +async def test_partial_final_chunk_is_padded_to_a_whole_tick(): + adapter = FakeAdapter([]) + scheduler = TickScheduler(adapter, bytes_per_tick=BYTES_PER_TICK) + scheduler.enqueue_utterance(b"\x02" * (BYTES_PER_TICK + 3)) + + await scheduler.run_tick() + await scheduler.run_tick() + + assert adapter.sent[1] == b"\x02" * 3 + b"\x00" * (BYTES_PER_TICK - 3) + assert scheduler.caller_is_speaking is False + + +async def test_caller_is_speaking_while_audio_remains_queued(): + scheduler = _scheduler([]) + scheduler.enqueue_utterance(b"\x02" * (BYTES_PER_TICK * 2)) + + assert scheduler.caller_is_speaking is True + await scheduler.run_tick() + assert scheduler.caller_is_speaking is True + await scheduler.run_tick() + assert scheduler.caller_is_speaking is False + + +async def test_simultaneous_speech_resets_both_counters_and_blocks_the_turn(): + scheduler = _scheduler([True]) + scheduler.enqueue_utterance(b"\x02" * BYTES_PER_TICK) + + await scheduler.run_tick() # both sides speak on tick 0 + + assert scheduler.may_take_turn() is False + + +async def test_tick_number_reaches_the_adapter_and_increments(): + adapter = FakeAdapter([]) + scheduler = TickScheduler(adapter, bytes_per_tick=BYTES_PER_TICK) + + for _ in range(3): + await scheduler.run_tick() + + assert adapter.received_ticks == [0, 1, 2] + assert scheduler.tick == 3 + + +async def test_consecutive_enqueues_drain_contiguously_with_no_silence_gap(): + adapter = FakeAdapter([]) + scheduler = TickScheduler(adapter, bytes_per_tick=BYTES_PER_TICK) + scheduler.enqueue_utterance(b"\x02" * BYTES_PER_TICK) + scheduler.enqueue_utterance(b"\x03" * BYTES_PER_TICK) + + await scheduler.run_tick() + await scheduler.run_tick() + + assert adapter.sent[0] == b"\x02" * BYTES_PER_TICK + assert adapter.sent[1] == b"\x03" * BYTES_PER_TICK + + +class RaisingAdapter(Adapter): + """Raises on a chosen tick to exercise the peek-then-commit failure path.""" + + def __init__(self, fail_on_tick: int) -> None: + self.fail_on_tick = fail_on_tick + + async def start(self) -> None: + pass + + async def run_tick(self, tick_number: int, outgoing_audio: bytes | None, *, barge_in: bool = False) -> TickResult: + if tick_number == self.fail_on_tick: + raise RuntimeError("adapter failure") + return TickResult( + tick_number=tick_number, + assistant_audio=b"\x00" * BYTES_PER_TICK, + assistant_audio_raw_bytes=0, + wall_clock_ms=tick_number, + ) + + async def stop(self) -> None: + pass + + +async def test_failed_adapter_call_leaves_queue_and_tick_unadvanced(): + adapter = RaisingAdapter(fail_on_tick=0) + scheduler = TickScheduler(adapter, bytes_per_tick=BYTES_PER_TICK) + utterance = b"\x02" * BYTES_PER_TICK + scheduler.enqueue_utterance(utterance) + + try: + await scheduler.run_tick() + except RuntimeError: + pass + + assert scheduler.tick == 0 + assert bytes(scheduler._playout) == utterance + + +async def test_check_tick_only_while_assistant_speaks_and_caller_is_silent(): + scheduler = _scheduler([True] * 40) + + # Tick 0..9 consumed; tick index 10 is the first multiple of the interval. + for _ in range(10): + await scheduler.run_tick() + + assert scheduler.is_check_tick() is True + + +async def test_not_a_check_tick_between_intervals(): + scheduler = _scheduler([True] * 40) + + for _ in range(11): + await scheduler.run_tick() + + assert scheduler.is_check_tick() is False + + +async def test_not_a_check_tick_when_the_assistant_is_silent(): + scheduler = _scheduler([True] + [False] * 40) + + for _ in range(10): + await scheduler.run_tick() + + assert scheduler.is_check_tick() is False + + +async def test_not_a_check_tick_while_the_caller_is_speaking(): + scheduler = _scheduler([True] * 40) + for _ in range(9): + await scheduler.run_tick() + scheduler.enqueue_utterance(b"\x02" * (BYTES_PER_TICK * 5)) + await scheduler.run_tick() + + assert scheduler.is_check_tick() is False + + +async def test_a_backchannel_does_not_consume_the_callers_turn(): + # A continuer earns no reply, so treating it as a turn deadlocks may_take_turn() + # until the inactivity timeout — 7/8 live conversations died this way. + scheduler = _scheduler([True] + [False] * 60) + await scheduler.run_tick() # tick 0: assistant greets + scheduler.enqueue_backchannel(b"\x02" * BYTES_PER_TICK) + await scheduler.run_tick() # caller says "mm-hmm" + + for _ in range(40): + await scheduler.run_tick() + + assert scheduler.may_take_turn() is True + + +async def test_a_real_utterance_still_consumes_the_turn(): + scheduler = _scheduler([True] + [False] * 60) + await scheduler.run_tick() + scheduler.enqueue_utterance(b"\x02" * BYTES_PER_TICK) + await scheduler.run_tick() + + for _ in range(40): + await scheduler.run_tick() + + assert scheduler.may_take_turn() is False + + +async def test_an_utterance_queued_after_a_backchannel_still_consumes_the_turn(): + scheduler = _scheduler([True] + [False] * 60) + await scheduler.run_tick() + scheduler.enqueue_backchannel(b"\x02" * BYTES_PER_TICK) + scheduler.enqueue_utterance(b"\x03" * BYTES_PER_TICK) + + for _ in range(40): + await scheduler.run_tick() + + assert scheduler.may_take_turn() is False + + +async def test_armed_barge_in_fires_on_the_tick_audio_reaches_the_wire(): + scheduler = _scheduler([True, True, True]) + adapter = scheduler._adapter + + await scheduler.run_tick() # silent tick: nothing on the wire yet + scheduler.arm_barge_in() + scheduler.enqueue_utterance(b"\x01" * BYTES_PER_TICK * 2) + await scheduler.run_tick() + await scheduler.run_tick() + + # Exactly the first tick that carried caller audio, and only that one. + assert adapter.barge_in_ticks == [1] + + +async def test_arming_a_barge_in_with_nothing_queued_does_not_fire_on_silence(): + scheduler = _scheduler([True, True]) + adapter = scheduler._adapter + + scheduler.arm_barge_in() + await scheduler.run_tick() + + assert adapter.barge_in_ticks == [] diff --git a/tests/unit/user_simulator/cascade/test_simulator.py b/tests/unit/user_simulator/cascade/test_simulator.py new file mode 100644 index 00000000..a7b824ca --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_simulator.py @@ -0,0 +1,893 @@ +from eva.user_simulator.cascade.phrases import load_phrases +from eva.user_simulator.cascade.simulator import CascadeUserSimulator, extract_turn, parse_turn_response + + +def _trace_sink(): + """DecisionLog that accumulates in memory and never writes.""" + from pathlib import Path + + from eva.user_simulator.cascade.decision_log import DecisionLog + + return DecisionLog(Path("unused-decision-trace.jsonl")) + + +def test_parse_turn_response_returns_the_spoken_line_unchanged(): + assert parse_turn_response("I need to reset my password.") == "I need to reset my password." + + +def test_parse_turn_response_returns_empty_for_a_toolcall_only_turn(): + # The model hangs up by calling end_call and says nothing; content is empty. + assert parse_turn_response("") == "" + + +def test_extract_turn_reads_a_plain_string_as_no_hangup(): + # LiteLLMClient returns a bare str when the model made no tool call. + assert extract_turn("Still here.") == ("Still here.", False) + + +def test_extract_turn_detects_the_end_call_tool(): + class _Fn: + name = "end_call" + + class _Call: + function = _Fn() + + class _Message: + content = "" + tool_calls = [_Call()] + + assert extract_turn(_Message()) == ("", True) + + +def test_extract_turn_ignores_an_unrelated_tool_call(): + class _Fn: + name = "something_else" + + class _Call: + function = _Fn() + + class _Message: + content = "Go on." + tool_calls = [_Call()] + + assert extract_turn(_Message()) == ("Go on.", False) + + +def test_outbound_perturbation_reaches_the_adapter(): + # background_noise / connection_degradation are applied per tick by RealtimeWSAdapter, + # so the cascade simulator no longer warns that it drops them. + assert not hasattr(CascadeUserSimulator, "_warn_unsupported_perturbation") + + +def _make_bare_simulator() -> CascadeUserSimulator: + """Build a CascadeUserSimulator without running __init__, for pure _messages() testing.""" + sim = object.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._build_prompt = lambda: "SYSTEM PROMPT" + sim._history = [] + return sim + + +def test_messages_flips_roles_so_the_user_simulator_llm_sees_its_own_lines_as_assistant(): + sim = _make_bare_simulator() + sim._history = [ + {"role": "assistant", "content": "What is your email?"}, + {"role": "user", "content": "It's jane@example.com."}, + ] + + messages = sim._messages() + + assert messages[0]["role"] == "system" + assert messages[1] == {"role": "user", "content": "What is your email?"} + assert messages[2] == {"role": "assistant", "content": "It's jane@example.com."} + + +class _FakeEventLogger: + def __init__(self) -> None: + self.events: list[tuple[str, dict]] = [] + + def log_event(self, name, data): + self.events.append((name, data)) + + +class _FakeScheduler: + tick = 7 + + +def _simulator_with_buffer(committed: str = "", in_flight: str = ""): + """Build a bare simulator with just the attributes _collect_heard_text touches.""" + from eva.user_simulator.cascade.stt import TranscriptBuffer + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + buffer = TranscriptBuffer() + buffer.committed, buffer.in_flight = committed, in_flight + sim._stt = type("_Stt", (), {"buffer": buffer})() + sim._ticks_awaiting_transcript = 0 + sim._missed_transcripts = 0 + sim.event_logger = _FakeEventLogger() + return sim + + +def test_committed_transcript_is_taken_immediately(): + sim = _simulator_with_buffer(committed="I can help with that.") + + assert sim._collect_heard_text(_FakeScheduler()) == ("I can help with that.", False) + + +def test_empty_buffer_waits_rather_than_generating_a_turn(): + # Finalization is not instant; the first empty read means "not ready yet". + sim = _simulator_with_buffer() + + assert sim._collect_heard_text(_FakeScheduler()) == ("", True) + + +def test_wait_expires_into_the_in_flight_partial(): + from eva.user_simulator.cascade.constants import TRANSCRIPT_WAIT_MS, ms_to_ticks + + sim = _simulator_with_buffer(in_flight="Please confirm your username") + for _ in range(ms_to_ticks(TRANSCRIPT_WAIT_MS)): + assert sim._collect_heard_text(_FakeScheduler()) == ("", True) + + assert sim._collect_heard_text(_FakeScheduler()) == ("Please confirm your username", False) + assert sim._stt.buffer.in_flight == "" + + +def test_the_wait_counter_resets_after_a_successful_read(): + sim = _simulator_with_buffer() + for _ in range(3): + sim._collect_heard_text(_FakeScheduler()) + sim._stt.buffer.committed = "Thanks, Marcus." + + sim._collect_heard_text(_FakeScheduler()) + + assert sim._ticks_awaiting_transcript == 0 + + +def _boundary_simulator(): + """Bare simulator exposing only what _log_audio_boundaries touches.""" + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim.event_logger = _FakeAudioEventLogger() + return sim + + +class _FakeAudioEventLogger: + def __init__(self) -> None: + self.calls: list[tuple[str, str, float]] = [] + + def log_audio_start(self, role, timestamp=None): + self.calls.append(("audio_start", role, timestamp)) + + def log_audio_end(self, role, timestamp=None): + self.calls.append(("audio_end", role, timestamp)) + + +class _Sched: + def __init__(self, spoke: bool) -> None: + self.caller_spoke_this_tick = spoke + + +def _tick(assistant_speech: bool, ms: int = 2000): + from eva.user_simulator.cascade.tick_result import TickResult + + return TickResult( + tick_number=0, + assistant_audio=b"\x00" * 8, + assistant_audio_raw_bytes=8 if assistant_speech else 0, + wall_clock_ms=ms, + ) + + +def test_caller_audio_start_is_logged_on_the_first_tick_of_playout(): + sim = _boundary_simulator() + + sim._log_audio_boundaries(_Sched(True), _tick(False), False, False) + + assert sim.event_logger.calls == [("audio_start", "simulated_user", 2.0)] + + +def test_caller_audio_end_is_logged_when_playout_stops(): + sim = _boundary_simulator() + + sim._log_audio_boundaries(_Sched(False), _tick(False), False, True) + + assert sim.event_logger.calls == [("audio_end", "simulated_user", 2.0)] + + +def test_no_event_while_the_caller_keeps_speaking(): + sim = _boundary_simulator() + + sim._log_audio_boundaries(_Sched(True), _tick(False), False, True) + + assert sim.event_logger.calls == [] + + +def test_assistant_boundaries_are_logged_too(): + # The metrics processor expects both roles, not just the user. + sim = _boundary_simulator() + + sim._log_audio_boundaries(_Sched(False), _tick(True), False, False) + sim._log_audio_boundaries(_Sched(False), _tick(False), True, False) + + assert sim.event_logger.calls == [ + ("audio_start", "assistant", 2.0), + ("audio_end", "assistant", 2.0), + ] + + +def test_timestamp_is_unix_seconds_not_milliseconds(): + # log_audio_* store the value as audio_timestamp, which metrics read as seconds. + sim = _boundary_simulator() + + sim._log_audio_boundaries(_Sched(True), _tick(False, ms=1786127928923), False, False) + + assert sim.event_logger.calls[0][2] == 1786127928.923 + + +def test_hearing_nothing_keeps_waiting_instead_of_speaking_into_the_void(): + # An assistant that never replies is an inactivity timeout, not a cue to talk again. + from eva.user_simulator.cascade.constants import TRANSCRIPT_WAIT_MS, ms_to_ticks + + sim = _simulator_with_buffer() + for _ in range(ms_to_ticks(TRANSCRIPT_WAIT_MS) + 3): + assert sim._collect_heard_text(_FakeScheduler()) == ("", True) + + +class _SilenceScheduler: + tick = 0 + assistant_has_spoken = True + + +def test_inactivity_ends_the_call_after_the_shared_two_minute_threshold(): + from eva.user_simulator.cascade.constants import INACTIVITY_TIMEOUT_MS, ms_to_ticks + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._ticks_assistant_silent = 0 + silent = _tick(False) + for _ in range(ms_to_ticks(INACTIVITY_TIMEOUT_MS)): + assert sim._assistant_is_inactive(_SilenceScheduler(), silent) is False + + assert sim._assistant_is_inactive(_SilenceScheduler(), silent) is True + + +def test_assistant_speech_resets_the_inactivity_counter(): + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._ticks_assistant_silent = 500 + + assert sim._assistant_is_inactive(_SilenceScheduler(), _tick(True)) is False + assert sim._ticks_assistant_silent == 0 + + +def test_inactivity_does_not_fire_before_the_assistant_ever_speaks(): + # The assistant opens the call; waiting for its greeting is not inactivity. + from eva.user_simulator.cascade.constants import INACTIVITY_TIMEOUT_MS, ms_to_ticks + + class _NeverSpoke: + tick = 0 + assistant_has_spoken = False + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._ticks_assistant_silent = 0 + for _ in range(ms_to_ticks(INACTIVITY_TIMEOUT_MS) + 5): + assert sim._assistant_is_inactive(_NeverSpoke(), _tick(False)) is False + + +def test_tick_counters_exist_without_manual_setup(): + # The unit tests build bare instances, so a counter initialised only inside __init__ + # would still pass them and then AttributeError on the first live tick. + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + + assert sim._ticks_assistant_silent == 0 + assert sim._ticks_awaiting_transcript == 0 + + +class RecordingScheduler: + """Captures what the simulator queues for playout.""" + + def __init__(self) -> None: + self.queued: list[bytes] = [] + self.backchannels: list[bytes] = [] + self.tick = 0 + + def enqueue_utterance(self, audio: bytes) -> None: + self.queued.append(audio) + + def enqueue_backchannel(self, audio: bytes) -> None: + self.backchannels.append(audio) + + +class StubCache: + """Phrase cache stand-in that always picks the first phrase.""" + + def choose(self, phrases): + return phrases[0] + + def get(self, phrase): + return b"CACHED" + + +async def test_backchannel_queues_cached_audio_without_synthesis(): + from eva.user_simulator.cascade import simulator as module + + scheduler = RecordingScheduler() + played = module.play_backchannel(scheduler, StubCache(), ["uh-huh", "mm-hmm"]) + + assert played == "uh-huh" + # Queued as a backchannel so it does not consume the caller's turn. + assert scheduler.backchannels == [b"CACHED"] + assert scheduler.queued == [] + + +def test_verdict_with_no_action_queues_nothing(): + from eva.user_simulator.cascade.decisions import ListenerVerdict + + verdict = ListenerVerdict(should_interrupt=False, should_backchannel=False) + + assert verdict.should_interrupt is False + assert verdict.should_backchannel is False + + +def test_a_barge_in_is_kept_while_the_assistant_is_mid_utterance(): + from eva.user_simulator.cascade.simulator import should_drop_interrupt + + assert should_drop_interrupt(assistant_still_speaking=True, same_assistant_turn=True) is False + + +def test_a_slow_barge_in_is_no_longer_dropped_for_being_slow(): + # Wall-clock slip is not staleness: the assistant is still talking, so the line still lands. + from eva.user_simulator.cascade.simulator import should_drop_interrupt + + assert should_drop_interrupt(assistant_still_speaking=True, same_assistant_turn=True) is False + + +def test_a_barge_in_is_dropped_once_the_assistant_stopped(): + # No longer an interruption — it would land as an ordinary reply. + from eva.user_simulator.cascade.simulator import should_drop_interrupt + + assert should_drop_interrupt(assistant_still_speaking=False, same_assistant_turn=True) is True + + +def test_a_barge_in_is_dropped_when_the_assistant_moved_to_a_later_turn(): + # "Still speaking" is satisfied by a *different* turn, which would land the line as a + # non-sequitur against speech the caller never reacted to. + from eva.user_simulator.cascade.simulator import should_drop_interrupt + + assert should_drop_interrupt(assistant_still_speaking=True, same_assistant_turn=False) is True + + +def test_extract_optional_line_reads_a_plain_line(): + from eva.user_simulator.cascade.simulator import extract_optional_line + + assert extract_optional_line("I wanted Friday, not Thursday.") == "I wanted Friday, not Thursday." + + +def test_extract_optional_line_strips_a_code_fence(): + from eva.user_simulator.cascade.simulator import extract_optional_line + + assert extract_optional_line("```\nI meant Friday.\n```") == "I meant Friday." + + +def test_extract_optional_line_is_empty_for_an_empty_reply(): + from eva.user_simulator.cascade.simulator import extract_optional_line + + assert extract_optional_line("") == "" + + +def test_extract_optional_line_reads_through_a_message_object(): + from eva.user_simulator.cascade.simulator import extract_optional_line + + class _Message: + content = "I meant Friday." + + assert extract_optional_line(_Message()) == "I meant Friday." + + +def test_extract_optional_line_rejects_a_refusal_style_non_answer(): + # The prompt allows the model to decline by returning NONE. + from eva.user_simulator.cascade.simulator import extract_optional_line + + assert extract_optional_line("NONE") == "" + + +async def test_relevance_gate_allows_a_still_relevant_candidate(): + from eva.user_simulator.cascade.simulator import candidate_is_relevant + from tests.unit.user_simulator.cascade.test_decisions import FakeLLM + + llm = FakeLLM(["YES"]) + + assert await candidate_is_relevant(llm, candidate="I wanted Friday.", heard="Booking Thursday now") is True + + +async def test_relevance_gate_rejects_a_stale_candidate(): + from eva.user_simulator.cascade.simulator import candidate_is_relevant + from tests.unit.user_simulator.cascade.test_decisions import FakeLLM + + llm = FakeLLM(["NO"]) + + assert await candidate_is_relevant(llm, candidate="I wanted Friday.", heard="What is your name?") is False + + +async def test_relevance_gate_fails_closed(): + from eva.user_simulator.cascade.simulator import candidate_is_relevant + from tests.unit.user_simulator.cascade.test_decisions import FakeLLM + + llm = FakeLLM(error=RuntimeError("down")) + + assert await candidate_is_relevant(llm, candidate="x", heard="y") is False + + +def test_slip_is_measured_on_the_wall_clock_not_the_tick_counter(): + # The tick counter cannot advance during _play_interruption's await: run_tick is + # only pumped by _run, so a tick-delta slip is structurally always zero. + from eva.user_simulator.cascade.simulator import interrupt_slip_ms + + assert interrupt_slip_ms(elapsed_s=1.0) == 1000 + assert interrupt_slip_ms(elapsed_s=0.0) == 0 + assert interrupt_slip_ms(elapsed_s=2.4) == 2400 + + +def test_slip_never_reports_negative_for_a_clock_hiccup(): + from eva.user_simulator.cascade.simulator import interrupt_slip_ms + + assert interrupt_slip_ms(elapsed_s=-0.5) == 0 + + +class _EndCallMessage: + """LLM reply that hangs up via the tool and says nothing.""" + + content = "" + + class _Fn: + name = "end_call" + + class _Call: + function = None + + def __init__(self) -> None: + call = self._Call() + call.function = self._Fn() + self.tool_calls = [call] + + +def _interrupting_simulator(message, framework="elevenlabs"): + """Bare simulator wired for _play_interruption only.""" + from eva.models.config import CascadeSimulatorConfig + from eva.user_simulator.cascade.stt import TranscriptBuffer + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._framework = framework + sim._config = CascadeSimulatorConfig(enable_interruptions=True) + sim._phrases = load_phrases("en") + sim._history = [] + sim._voice_id = "voice-f" + sim._build_prompt = lambda: "SYSTEM PROMPT" + sim.event_logger = _FakeEventLogger() + sim._phrase_cache = StubCache() + sim._record_audio = lambda *a, **k: None + sim._on_user_speaks = lambda *a, **k: None + sim._on_assistant_speaks = lambda *a, **k: None + sim.ended = [] + sim._on_conversation_end = sim.ended.append + buffer = TranscriptBuffer() + buffer.committed = "Your account is unlocked." + sim._stt = type("_Stt", (), {"buffer": buffer})() + + class _Llm: + async def complete(self, messages, tools=None): + return message, {} + + class _Tts: + async def stream(self, text, *, voice_id): + yield text.encode() + + sim._llm, sim._tts = _Llm(), _Tts() + return sim + + +class _InterruptScheduler: + tick = 40 + assistant_is_speaking = True + + def __init__(self) -> None: + self.queued: list[bytes] = [] + self.barge_in_armed = False + + def arm_barge_in(self) -> None: + self.barge_in_armed = True + + def enqueue_utterance(self, audio: bytes) -> None: + self.queued.append(audio) + + +async def test_a_hangup_during_an_interruption_ends_the_call(): + # Discarding end_call here stranded the caller: it emitted nothing and the + # conversation ran on to the inactivity timeout, looping the assistant. + sim = _interrupting_simulator(_EndCallMessage()) + + hung_up = await sim._play_interruption(_InterruptScheduler()) + + assert hung_up is True + assert sim.ended == ["goodbye"] + + +async def test_a_hangup_is_never_dropped_as_stale(): + sim = _interrupting_simulator(_EndCallMessage()) + + class _StoppedScheduler(_InterruptScheduler): + assistant_is_speaking = False # would normally drop the interruption + + assert await sim._play_interruption(_StoppedScheduler()) is True + assert sim.ended == ["goodbye"] + + +async def test_an_ordinary_interruption_does_not_end_the_call(): + sim = _interrupting_simulator("It says my account is locked out.") + + assert await sim._play_interruption(_InterruptScheduler()) is False + assert sim.ended == [] + + +async def test_only_one_interruption_fires_per_assistant_turn(): + # The eligibility flag is cleared on firing, so a second check in the same + # assistant turn is offered allow_interrupt=False and cannot barge in again. + from eva.models.config import CascadeSimulatorConfig + from eva.user_simulator.cascade.decisions import ListenerVerdict + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._config = CascadeSimulatorConfig(enable_interruptions=True) + sim._phrase_cache = StubCache() + sim._may_interrupt_this_turn = True + sim.plays = 0 + offered = [] + + class _Decisions: + async def evaluate(self, text, *, allow_interrupt, allow_backchannel): + offered.append(allow_interrupt) + return ListenerVerdict(should_interrupt=allow_interrupt, should_backchannel=False) + + async def _play(scheduler): + sim.plays += 1 + return False + + sim._decisions = _Decisions() + sim._play_interruption = _play + # The transcript has to grow between the two checks, or the unchanged-transcript + # gate skips the second one before eligibility is ever consulted. + from eva.user_simulator.cascade.stt import TranscriptBuffer + + buffer = TranscriptBuffer() + sim._stt = type("_Stt", (), {"buffer": buffer})() + + buffer.committed = "Let me pull up" + await sim._run_checks(_InterruptScheduler()) + buffer.committed = "Let me pull up your account." + await sim._run_checks(_InterruptScheduler()) + + assert offered == [True, False] + assert sim.plays == 1 + + +def test_a_short_pause_inside_one_assistant_turn_is_not_a_new_turn(): + # A single quiet tick used to re-arm the interruption cap mid-utterance. + from eva.user_simulator.cascade.simulator import is_new_assistant_turn + + assert is_new_assistant_turn(ticks_silent_before=1) is False + assert is_new_assistant_turn(ticks_silent_before=4) is False + + +def test_a_sustained_gap_starts_a_new_assistant_turn(): + from eva.user_simulator.cascade.constants import WAIT_TO_RESPOND_OTHER_MS, ms_to_ticks + from eva.user_simulator.cascade.simulator import is_new_assistant_turn + + assert is_new_assistant_turn(ticks_silent_before=ms_to_ticks(WAIT_TO_RESPOND_OTHER_MS)) is True + + +def test_inactivity_measures_contiguous_silence_not_cumulative(): + # The reset was unreachable in the live loop, so scattered quiet ticks accumulated + # and killed healthy calls once they happened to total two minutes. + from eva.user_simulator.cascade.constants import INACTIVITY_TIMEOUT_MS, ms_to_ticks + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._ticks_assistant_silent = 0 + limit = ms_to_ticks(INACTIVITY_TIMEOUT_MS) + + # Almost time out, then the assistant speaks once, then go quiet again. + for _ in range(limit): + assert sim._assistant_is_inactive(_SilenceScheduler(), _tick(False)) is False + assert sim._assistant_is_inactive(_SilenceScheduler(), _tick(True)) is False + for _ in range(limit): + assert sim._assistant_is_inactive(_SilenceScheduler(), _tick(False)) is False + + assert sim._assistant_is_inactive(_SilenceScheduler(), _tick(False)) is True + + +async def test_a_dropped_interruption_emits_no_audio_at_all(): + # Emitting the opener up front meant a stale drop left it orphaned on the wire. + sim = _interrupting_simulator("Active Directory.") + + class _StaleScheduler(_InterruptScheduler): + assistant_is_speaking = False # forces should_drop_interrupt + + scheduler = _StaleScheduler() + assert await sim._play_interruption(scheduler) is False + assert scheduler.queued == [] + + +async def test_a_hangup_during_an_interruption_emits_no_opener(): + sim = _interrupting_simulator(_EndCallMessage()) + scheduler = _InterruptScheduler() + + assert await sim._play_interruption(scheduler) is True + assert scheduler.queued == [] + assert sim.ended == ["goodbye"] + + +async def test_a_kept_interruption_speaks_the_opener_then_the_content(): + sim = _interrupting_simulator("Active Directory.") + scheduler = _InterruptScheduler() + + assert await sim._play_interruption(scheduler) is False + + assert scheduler.queued[0] == b"CACHED" + assert b"Active Directory." in b"".join(scheduler.queued[1:]) + # A tick-driven transport needs to know this audio cuts the assistant off. + assert scheduler.barge_in_armed is True + + +def test_openai_realtime_uses_the_tick_driven_adapter(): + from eva.user_simulator.cascade.adapter.tick_driven import TickDrivenAdapter + from eva.user_simulator.cascade.simulator import adapter_class_for_framework + + assert adapter_class_for_framework("openai_realtime") is TickDrivenAdapter + + +def test_pipecat_stays_on_the_real_time_adapter(): + # Pipecat owns its own clock; freezing it is not possible. + from eva.user_simulator.cascade.adapter.realtime_ws import RealtimeWSAdapter + from eva.user_simulator.cascade.simulator import adapter_class_for_framework + + assert adapter_class_for_framework("pipecat") is RealtimeWSAdapter + + +def test_elevenlabs_stays_on_the_real_time_adapter(): + from eva.user_simulator.cascade.adapter.realtime_ws import RealtimeWSAdapter + from eva.user_simulator.cascade.simulator import adapter_class_for_framework + + assert adapter_class_for_framework("elevenlabs") is RealtimeWSAdapter + + +def test_unported_frameworks_default_to_the_real_time_adapter(): + from eva.user_simulator.cascade.adapter.realtime_ws import RealtimeWSAdapter + from eva.user_simulator.cascade.simulator import adapter_class_for_framework + + assert adapter_class_for_framework("gemini_live") is RealtimeWSAdapter + + +async def test_the_incomplete_marker_never_reaches_the_conversation_history(): + sim = _interrupting_simulator("Active Directory.") + sim._stt.buffer.apply_partial("and I still nee") + + await sim._play_interruption(_InterruptScheduler()) + + heard = [m["content"] for m in sim._history if m["role"] == "assistant"] + assert heard == ["Your account is unlocked. and I still nee"] + + +async def test_a_slow_barge_in_survives_on_every_transport(): + # Slip is now reported, not enforced: the assistant is mid-utterance, so the line lands. + from eva.user_simulator.cascade import simulator as module + + for framework in ("openai_realtime", "pipecat"): + sim = _interrupting_simulator("Active Directory.", framework=framework) + scheduler = _InterruptScheduler() + + original = module.interrupt_slip_ms + module.interrupt_slip_ms = lambda *, elapsed_s: 2400 + try: + assert await sim._play_interruption(scheduler) is False + finally: + module.interrupt_slip_ms = original + + assert scheduler.queued != [], framework + + +async def test_a_barge_in_is_abandoned_when_a_new_assistant_turn_started_meanwhile(): + # The assistant finished the utterance we reacted to and began another one while the + # line was being generated; firing now would answer speech the caller never heard. + sim = _interrupting_simulator("Active Directory.", framework="pipecat") + scheduler = _InterruptScheduler() + + original_complete = sim._llm.complete + + async def _complete_then_new_turn(messages, tools=None): + sim._assistant_turn_index += 1 + return await original_complete(messages, tools) + + sim._llm.complete = _complete_then_new_turn + + assert await sim._play_interruption(scheduler) is False + assert scheduler.queued == [] + + +def test_provider_stall_is_a_distinct_terminal_reason_from_inactivity(): + # inactivity_timeout is a legitimate end the metrics treat as definitive when the + # user spoke last; a stalled peer is an invalid record the runner should retry. The + # two must not share a reason, or a dead provider scores as a finished conversation. + from eva.user_simulator.cascade.tick_result import TickResult + + stalled = TickResult( + tick_number=9, + assistant_audio=b"\x00" * 8, + assistant_audio_raw_bytes=0, + wall_clock_ms=0, + provider_stalled=True, + ) + + assert stalled.provider_stalled is True + assert ( + TickResult(tick_number=9, assistant_audio=b"", assistant_audio_raw_bytes=0, wall_clock_ms=0).provider_stalled + is False + ) + + +class _CountingDecisions: + """ListenerDecisions stand-in that records every evaluate() it is asked to run.""" + + def __init__(self) -> None: + self.calls: list[str] = [] + + async def evaluate(self, heard, *, allow_interrupt, allow_backchannel): + from eva.user_simulator.cascade.decisions import ListenerVerdict + + self.calls.append(heard) + self.last_allow_backchannel = allow_backchannel + return ListenerVerdict(should_interrupt=False, should_backchannel=False) + + +def _checking_simulator(): + from eva.models.config import CascadeSimulatorConfig + from eva.user_simulator.cascade.stt import TranscriptBuffer + + sim = CascadeUserSimulator.__new__(CascadeUserSimulator) + sim._decision_log = _trace_sink() + sim._config = CascadeSimulatorConfig(enable_interruptions=True, enable_backchannel=True) + sim._phrases = load_phrases("en") + sim._decisions = _CountingDecisions() + sim._phrase_cache = StubCache() + sim._may_interrupt_this_turn = True + sim._last_checked_text = "" + sim.event_logger = _FakeEventLogger() + sim._record_audio = lambda *a, **k: None + buffer = TranscriptBuffer() + sim._stt = type("_Stt", (), {"buffer": buffer})() + return sim, buffer + + +async def test_an_unchanged_transcript_does_not_re_ask_the_judges(): + # The checks fire on a timer but read only the transcript, so a tick where it has + # not moved re-asks a question already answered — ~101 checks x 2 calls per call. + sim, buffer = _checking_simulator() + buffer.committed = "Let me pull up your account." + + await sim._run_checks(_FakeScheduler()) + await sim._run_checks(_FakeScheduler()) + await sim._run_checks(_FakeScheduler()) + + assert sim._decisions.calls == ["Let me pull up your account."] + + +async def test_a_grown_transcript_is_judged_again(): + sim, buffer = _checking_simulator() + buffer.committed = "Let me pull up" + + await sim._run_checks(_FakeScheduler()) + buffer.committed = "Let me pull up your account." + await sim._run_checks(_FakeScheduler()) + + assert len(sim._decisions.calls) == 2 + + +async def test_an_empty_transcript_is_never_judged(): + sim, _buffer = _checking_simulator() + + await sim._run_checks(_FakeScheduler()) + + assert sim._decisions.calls == [] + + +async def test_backchannelling_survives_a_barge_in_in_the_same_turn(): + # Interrupting and later humming along are not mutually exclusive for a real + # listener; the per-turn cap is about not talking over the assistant twice, not + # about going silent for the rest of the turn. + sim, buffer = _checking_simulator() + sim._may_interrupt_this_turn = False # this turn has already barged in + buffer.committed = "Let me pull up your account." + + await sim._run_checks(_FakeScheduler()) + + assert sim._decisions.last_allow_backchannel is True + + +async def test_backchannelling_is_off_only_when_the_config_disables_it(): + from eva.models.config import CascadeSimulatorConfig + + sim, buffer = _checking_simulator() + sim._config = CascadeSimulatorConfig(enable_interruptions=True, enable_backchannel=False) + buffer.committed = "Let me pull up your account." + + await sim._run_checks(_FakeScheduler()) + + assert sim._decisions.last_allow_backchannel is False + + +async def test_a_transcript_already_judged_in_an_earlier_turn_is_judged_again(): + # take_committed() runs between turns, so current_text() can return a string this + # gate has already seen. Without clearing on a turn boundary that later turn's + # check would be skipped as "unchanged" and the barge-in never considered. + sim, buffer = _checking_simulator() + buffer.committed = "Anything else I can help with?" + await sim._run_checks(_FakeScheduler()) + assert len(sim._decisions.calls) == 1 + + # The ordinary turn path consumes the transcript, then the assistant says the same + # thing again a turn later — a real pattern for closing questions. + buffer.take_committed() + sim._last_checked_text = "" # what the new-turn branch in _run does + buffer.committed = "Anything else I can help with?" + + await sim._run_checks(_FakeScheduler()) + + assert len(sim._decisions.calls) == 2 + + +async def test_a_skipped_check_writes_no_listener_check_row(tmp_path): + # The trace exists to separate "the model said NO" from "the check never ran". A row + # for a check that was gated out before reaching the judges would blur exactly that. + from eva.user_simulator.cascade.decision_log import DecisionLog + + sim, buffer = _checking_simulator() + sim._decision_log = DecisionLog(tmp_path / "trace.jsonl") + buffer.committed = "Let me pull up your account." + + await sim._run_checks(_FakeScheduler()) + await sim._run_checks(_FakeScheduler()) + await sim._run_checks(_FakeScheduler()) + + assert sim._decisions.calls == ["Let me pull up your account."] + assert sim._decision_log._counts.get("listener_check") == 1 + + +async def test_the_tick_trace_records_whether_the_transcript_moved(tmp_path): + # A check tick with no listener_check row is otherwise ambiguous between "nothing + # new was heard" and "the check ran and failed". + from eva.user_simulator.cascade.decision_log import DecisionLog + + sim, buffer = _checking_simulator() + rows = [] + sim._decision_log = DecisionLog(tmp_path / "trace.jsonl") + sim._decision_log.log = lambda kind, **fields: rows.append((kind, fields)) + sim._ticks_assistant_silent = 0 + sim._ticks_since_assistant_started = 3 + sim._candidate_audio = b"" + buffer.committed = "Let me pull up your account." + + sim._log_tick_state(_TickTraceScheduler(), _tick(True), is_check_tick=True) + sim._last_checked_text = buffer.current_text() + sim._log_tick_state(_TickTraceScheduler(), _tick(True), is_check_tick=True) + + assert rows[0][1]["transcript_moved"] is True + assert rows[1][1]["transcript_moved"] is False + + +class _TickTraceScheduler: + tick = 12 + caller_is_speaking = False + caller_spoke_this_tick = False diff --git a/tests/unit/user_simulator/cascade/test_stt.py b/tests/unit/user_simulator/cascade/test_stt.py new file mode 100644 index 00000000..357b0fa2 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_stt.py @@ -0,0 +1,70 @@ +from eva.user_simulator.cascade.stt import TranscriptBuffer + + +def test_partial_updates_replace_the_in_flight_text(): + buffer = TranscriptBuffer() + + buffer.apply_partial("Let me check") + buffer.apply_partial("Let me check that for") + + assert buffer.in_flight == "Let me check that for" + assert buffer.committed == "" + + +def test_commit_appends_and_clears_the_partial(): + buffer = TranscriptBuffer() + buffer.apply_partial("Let me check that") + + buffer.commit("Let me check that for you.") + + assert buffer.committed == "Let me check that for you." + assert buffer.in_flight == "" + + +def test_successive_commits_accumulate_with_spaces(): + buffer = TranscriptBuffer() + + buffer.commit("First sentence.") + buffer.commit("Second sentence.") + + assert buffer.committed == "First sentence. Second sentence." + + +def test_take_committed_drains_the_buffer(): + buffer = TranscriptBuffer() + buffer.commit("All done.") + + assert buffer.take_committed() == "All done." + assert buffer.committed == "" + + +def test_current_text_marks_the_in_flight_partial_as_incomplete(): + buffer = TranscriptBuffer() + buffer.commit("I found your order.") + buffer.apply_partial("It includes a keyboa") + + assert buffer.current_text() == "I found your order. It includes a keyboa [CURRENTLY SPEAKING, INCOMPLETE]" + + +def test_current_text_omits_the_marker_when_nothing_is_in_flight(): + buffer = TranscriptBuffer() + buffer.commit("I found your order.") + + assert buffer.current_text() == "I found your order." + + +def test_current_text_does_not_consume_the_committed_text(): + buffer = TranscriptBuffer() + buffer.commit("hello") + + buffer.current_text() + + assert buffer.take_committed() == "hello" + + +def test_heard_text_never_carries_the_prompt_marker_into_the_transcript(): + buffer = TranscriptBuffer() + buffer.commit("I found your order.") + buffer.apply_partial("It includes a keyboa") + + assert buffer.heard_text() == "I found your order. It includes a keyboa" diff --git a/tests/unit/user_simulator/cascade/test_stt_livekit.py b/tests/unit/user_simulator/cascade/test_stt_livekit.py new file mode 100644 index 00000000..cb68f2b7 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_stt_livekit.py @@ -0,0 +1,43 @@ +import pytest + +from eva.user_simulator.cascade.stt import TranscriptBuffer +from eva.user_simulator.cascade.stt_livekit import LiveKitStreamingSTT, build_livekit_stt + + +def test_unknown_provider_is_rejected_with_a_clear_message(): + with pytest.raises(ValueError, match="Unsupported caller STT provider"): + build_livekit_stt("nope", {}) + + +def test_elevenlabs_provider_defaults_to_the_realtime_scribe_model(): + # Only scribe_v2_realtime streams interim transcripts and honours flush(). + stt = LiveKitStreamingSTT("elevenlabs", {"api_key": "k"}) + + assert stt.model == "scribe_v2_realtime" + + +def test_explicit_model_overrides_the_default(): + stt = LiveKitStreamingSTT("elevenlabs", {"api_key": "k", "model": "scribe_v2"}) + + assert stt.model == "scribe_v2" + + +def test_buffer_starts_empty_and_is_a_transcript_buffer(): + stt = LiveKitStreamingSTT("elevenlabs", {"api_key": "k"}) + + assert isinstance(stt.buffer, TranscriptBuffer) + assert stt.buffer.committed == "" + assert stt.buffer.in_flight == "" + + +async def test_feed_before_start_is_a_noop_rather_than_an_error(): + stt = LiveKitStreamingSTT("elevenlabs", {"api_key": "k"}) + + await stt.feed(b"\x00" * 320) + + +async def test_stop_is_safe_before_start_and_twice(): + stt = LiveKitStreamingSTT("elevenlabs", {"api_key": "k"}) + + await stt.stop() + await stt.stop() diff --git a/tests/unit/user_simulator/cascade/test_tick_driven_adapter.py b/tests/unit/user_simulator/cascade/test_tick_driven_adapter.py new file mode 100644 index 00000000..8e9a43ac --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_tick_driven_adapter.py @@ -0,0 +1,207 @@ +import asyncio +import json +import time + +from eva.user_simulator.cascade.adapter.tick_driven import ( + MAX_INACTIVE_SECONDS, + QUIET_TICK_GRACE_S, + TickDrivenAdapter, +) +from tests.unit.user_simulator.cascade.test_realtime_ws_adapter import ( + BYTES_PER_TICK, + FakeWebSocket, + _media_frame, + _settle, +) + + +def _media(ws: FakeWebSocket) -> list[dict]: + return [json.loads(m) for m in ws.sent if json.loads(m).get("event") == "media"] + + +async def test_outbound_audio_is_sent_without_pacing_sleeps(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + # Assistant audio already buffered, so the tick has no reason to wait for any. + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + + started = time.monotonic() + await adapter.run_tick(0, b"\x00" * BYTES_PER_TICK) + + # The real-time adapter would spend ~200ms pacing the outbound frames. + assert time.monotonic() - started < 0.05 + assert len(_media(ws)) == 10 + await adapter.stop() + + +async def test_burst_of_provider_audio_releases_one_tick_at_a_time(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + # 1 second of assistant audio arrives at once: 8000 mulaw bytes -> 32000 PCM bytes. + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + + results = [await adapter.run_tick(tick, None) for tick in range(5)] + + # The resampler's filter state leaves the last tick a couple of samples short; + # what matters is that no tick releases more than one tick's worth. + raw = [r.assistant_audio_raw_bytes for r in results] + assert raw[:4] == [BYTES_PER_TICK] * 4 + assert 0 < raw[4] <= BYTES_PER_TICK + # Short ticks are silence-padded, so every released chunk is exactly tick-sized. + assert all(len(r.assistant_audio) == BYTES_PER_TICK for r in results) + await adapter.stop() + + +async def test_a_silent_tick_still_puts_a_full_tick_of_silence_on_the_wire(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + await adapter.run_tick(0, None) + + # The provider's VAD ends the caller's turn on received silence. A tick that + # emitted nothing would starve it and the assistant would never reply. + assert len(_media(ws)) == 10 + await adapter.stop() + + +async def test_played_position_tracks_released_ticks(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + + for tick in range(3): + await adapter.run_tick(tick, None) + + assert adapter.played_ms == 600 + await adapter.stop() + + +async def test_a_tick_with_no_assistant_audio_does_not_advance_the_played_position(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + for tick in range(3): + await adapter.run_tick(tick, None) + + assert adapter.played_ms == 0 + await adapter.stop() + + +async def test_ticks_with_audio_already_buffered_do_not_wait_out_the_tick_duration(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + # 1s of audio arrives at once; releasing it must not take 1s of wall time. + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + + started = time.monotonic() + for tick in range(4): + await adapter.run_tick(tick, None) + + # The real-time adapter enforces a 200ms floor per tick; this one must not. + assert time.monotonic() - started < 0.05 + await adapter.stop() + + +async def test_a_tick_waits_out_the_grace_before_calling_the_assistant_silent(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + started = time.monotonic() + result = await adapter.run_tick(0, None) + + # Without this bound the tick loop spins through its whole budget instantly and + # the call ends at tick zero having heard nothing. + assert result.assistant_audio_raw_bytes == 0 + assert time.monotonic() - started >= QUIET_TICK_GRACE_S + await adapter.stop() + + +async def test_a_tick_returns_as_soon_as_audio_arrives_mid_grace(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + + async def _deliver_late() -> None: + await asyncio.sleep(QUIET_TICK_GRACE_S / 4) + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + + asyncio.create_task(_deliver_late()) + started = time.monotonic() + result = await adapter.run_tick(0, None) + + assert result.assistant_audio_raw_bytes == BYTES_PER_TICK + assert time.monotonic() - started < QUIET_TICK_GRACE_S + await adapter.stop() + + +async def test_barge_in_reports_the_played_position_not_the_received_position(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + # 1s of audio arrives at once but only 3 ticks (600ms) get released. + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + for tick in range(3): + await adapter.run_tick(tick, None) + + result = await adapter.run_tick(3, b"\x00" * BYTES_PER_TICK, barge_in=True) + + assert result.interruption_audio_start_ms == 600 + truncate = [json.loads(m) for m in ws.sent if json.loads(m).get("event") == "truncate"] + assert truncate and truncate[0]["audio_end_ms"] == 600 + await adapter.stop() + + +async def test_barge_in_discards_audio_the_caller_never_heard(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + + result = await adapter.run_tick(0, b"\x00" * BYTES_PER_TICK, barge_in=True) + + # The buffered second of assistant audio is audio the caller cut off. + assert result.assistant_audio_raw_bytes == 0 + assert adapter.played_ms == 0 + await adapter.stop() + + +async def test_a_stalled_provider_is_reported_not_raised(): + # Raising aborted the tick loop from under the simulator, so the record ended on the + # generic "error" reason and the terminal state the runner reads was never written. + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + adapter._last_inbound_monotonic = time.monotonic() - (MAX_INACTIVE_SECONDS + 1) + + result = await adapter.run_tick(0, None) + + assert result.provider_stalled is True + await adapter.stop() + + +async def test_a_live_provider_is_never_reported_as_stalled(): + ws = FakeWebSocket() + adapter = TickDrivenAdapter(websocket=ws, conversation_id="c1", bytes_per_tick=BYTES_PER_TICK) + await adapter.start() + adapter._last_inbound_monotonic = time.monotonic() - (MAX_INACTIVE_SECONDS + 1) + await ws.inbound.put(_media_frame(b"\xff" * 8000)) + await _settle() + + result = await adapter.run_tick(0, None) + + # Audio arrived this tick, so the liveness clock resets rather than firing. + assert result.provider_stalled is False + await adapter.stop() diff --git a/tests/unit/user_simulator/cascade/test_tick_result.py b/tests/unit/user_simulator/cascade/test_tick_result.py new file mode 100644 index 00000000..57a338ed --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_tick_result.py @@ -0,0 +1,74 @@ +import pytest + +from eva.user_simulator.cascade.tick_result import TickResult, played_audio_ms, split_tick_audio + + +@pytest.mark.parametrize("input_length", [0, 1, 7, 8, 9, 17]) +def test_split_chunk_is_always_bytes_per_tick(input_length): + chunk, overflow = split_tick_audio(b"\x01" * input_length, bytes_per_tick=8) + + assert len(chunk) == 8 + assert len(overflow) == max(0, input_length - 8) + + +def test_split_pads_short_audio_with_silence(): + chunk, overflow = split_tick_audio(b"\x01\x02", bytes_per_tick=8) + + assert chunk == b"\x01\x02" + b"\x00" * 6 + assert overflow == b"" + + +def test_split_carries_overflow_to_next_tick(): + chunk, overflow = split_tick_audio(b"\x01" * 12, bytes_per_tick=8) + + assert chunk == b"\x01" * 8 + assert overflow == b"\x01" * 4 + + +def test_split_of_empty_audio_is_all_silence(): + chunk, overflow = split_tick_audio(b"", bytes_per_tick=8) + + assert chunk == b"\x00" * 8 + assert overflow == b"" + + +def test_has_assistant_speech_is_false_for_padded_silence(): + result = TickResult(tick_number=3, assistant_audio=b"\x00" * 8, assistant_audio_raw_bytes=0, wall_clock_ms=1) + + assert result.has_assistant_speech is False + + +def test_has_assistant_speech_is_true_when_real_audio_arrived(): + result = TickResult(tick_number=3, assistant_audio=b"\x01" * 8, assistant_audio_raw_bytes=8, wall_clock_ms=1) + + assert result.has_assistant_speech is True + + +def test_played_position_reflects_released_ticks_not_received_bytes(): + # 12 ticks released at 200ms each, regardless of how much arrived early. + assert played_audio_ms(ticks_released=12) == 2400 + + +def test_played_position_is_zero_before_anything_is_released(): + assert played_audio_ms(ticks_released=0) == 0 + + +def test_truncation_defaults_are_inert(): + result = TickResult(tick_number=1, assistant_audio=b"\x00" * 8, assistant_audio_raw_bytes=0, wall_clock_ms=0) + + assert result.skip_item_id is None + assert result.interruption_audio_start_ms is None + + +def test_truncation_fields_carry_the_played_position(): + result = TickResult( + tick_number=1, + assistant_audio=b"\x00" * 8, + assistant_audio_raw_bytes=0, + wall_clock_ms=0, + skip_item_id="item_42", + interruption_audio_start_ms=2400, + ) + + assert result.skip_item_id == "item_42" + assert result.interruption_audio_start_ms == 2400 diff --git a/tests/unit/user_simulator/cascade/test_tts.py b/tests/unit/user_simulator/cascade/test_tts.py new file mode 100644 index 00000000..242bd595 --- /dev/null +++ b/tests/unit/user_simulator/cascade/test_tts.py @@ -0,0 +1,70 @@ +import pytest + +from eva.user_simulator.cascade.tts import DEFAULT_FEMALE_VOICE, DEFAULT_MALE_VOICE, CartesiaTTS + + +def test_voice_id_selected_for_female_persona(): + tts = CartesiaTTS({"model": "sonic-3.5", "female_voice": "voice-f", "male_voice": "voice-m"}) + + assert tts.voice_for_persona({"user_persona_id": 1}) == "voice-f" + + +def test_voice_id_selected_for_male_persona(): + tts = CartesiaTTS({"model": "sonic-3.5", "female_voice": "voice-f", "male_voice": "voice-m"}) + + assert tts.voice_for_persona({"user_persona_id": 2}) == "voice-m" + + +def test_unknown_persona_falls_back_to_female_voice(): + tts = CartesiaTTS({"model": "sonic-3.5", "female_voice": "voice-f", "male_voice": "voice-m"}) + + assert tts.voice_for_persona({}) == "voice-f" + + +async def test_empty_text_synthesizes_to_no_audio(): + tts = CartesiaTTS({"model": "sonic-3.5", "api_key": "k"}) + + assert await tts.synthesize("", voice_id="voice-f") == b"" + + +async def test_missing_api_key_raises_a_clear_error(monkeypatch): + monkeypatch.delenv("CARTESIA_API_KEY", raising=False) + tts = CartesiaTTS({"model": "sonic-3.5"}) + + with pytest.raises(ValueError, match="Cartesia API key"): + await tts.synthesize("hello", voice_id="voice-f") + + +async def test_stream_yields_nothing_for_empty_text(): + tts = CartesiaTTS({"model": "sonic-3.5", "api_key": "k"}) + + chunks = [chunk async for chunk in tts.stream("", voice_id="voice-f")] + + assert chunks == [] + + +async def test_synthesize_concatenates_the_stream(monkeypatch): + tts = CartesiaTTS({"model": "sonic-3.5", "api_key": "k"}) + + async def fake_stream(text, *, voice_id): + for piece in (b"ab", b"cd"): + yield piece + + monkeypatch.setattr(tts, "stream", fake_stream) + + assert await tts.synthesize("hello", voice_id="voice-f") == b"abcd" + + +def test_the_two_default_voices_are_actually_different(): + # They were the same id, so voice_for_persona returned one voice for every + # persona and the gender scheme it documents was inert. + assert DEFAULT_FEMALE_VOICE != DEFAULT_MALE_VOICE + + +def test_personas_of_different_genders_get_different_voices(): + tts = CartesiaTTS({"api_key": "k"}) + + female = tts.voice_for_persona({"user_persona_id": 1}) + male = tts.voice_for_persona({"user_persona_id": 2}) + + assert female != male diff --git a/tests/unit/user_simulator/test_factory.py b/tests/unit/user_simulator/test_factory.py index d47f1abe..74fdb4b9 100644 --- a/tests/unit/user_simulator/test_factory.py +++ b/tests/unit/user_simulator/test_factory.py @@ -43,3 +43,26 @@ def test_factory_selects_openai_realtime(tmp_path): assert isinstance(simulator, OpenAIRealtimeUserSimulator) assert simulator.caller_model == "gpt-realtime-1.5" + + +def test_factory_selects_cascade(tmp_path): + from eva.models.config import CascadeSimulatorConfig + from eva.user_simulator.cascade.simulator import CascadeUserSimulator + from eva.utils import router + + router.init( + model_list=[ + { + "model_name": "gpt-5.5", + "litellm_params": {"model": "openai/gpt-5.5", "api_key": "test-key"}, + } + ] + ) + try: + config = CascadeSimulatorConfig() + simulator = create_user_simulator(config, **_kwargs(tmp_path)) + + assert isinstance(simulator, CascadeUserSimulator) + assert simulator.provider == "cascade" + finally: + router.reset() diff --git a/uv.lock b/uv.lock index 4eae4cdb..c8281984 100644 --- a/uv.lock +++ b/uv.lock @@ -294,6 +294,23 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7e/16/fbe8e1e185a45042f7cd3a282def5bb8d95bb69ab9e9ef6a5368aa17e426/audioread-3.1.0-py3-none-any.whl", hash = "sha256:b30d1df6c5d3de5dcef0fb0e256f6ea17bdcf5f979408df0297d8a408e2971b4", size = 23143, upload-time = "2025-10-26T19:44:12.016Z" }, ] +[[package]] +name = "av" +version = "18.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ae/a4/570a5a35c8638aba01e739925846c35fdd6b0756a15526766d0a4dd3b7df/av-18.0.0.tar.gz", hash = "sha256:4ef7e72c3d3a872584a1215173b16e0226811037f40dcdbf75992631098df1ba", size = 4340222, upload-time = "2026-07-02T06:37:58.907Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/18/4a/9e3463df030e063d757fa12f0f39be6541b45b06b5bad48c2ce361b924bf/av-18.0.0-cp311-abi3-macosx_11_0_x86_64.whl", hash = "sha256:149289d40e732a6e49c9530bc245b49d9964cfd1c8c9e06778703b7d5bba6b25", size = 22499354, upload-time = "2026-07-02T06:36:58.751Z" }, + { url = "https://files.pythonhosted.org/packages/77/b3/2576a44b4f39c7462ced4c17fec04c756f7b0f3c5cb940d124173e417d6a/av-18.0.0-cp311-abi3-macosx_14_0_arm64.whl", hash = "sha256:35274c20d2ad3b4774fe632bcef2e34af79858ddf899352339cc3babbc13a484", size = 18175248, upload-time = "2026-07-02T06:37:01.741Z" }, + { url = "https://files.pythonhosted.org/packages/84/74/6732f17b96dc23fd23b876b2805435855abdc8a3b397142be4e581165de8/av-18.0.0-cp311-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:4d683b7747a0ba9222b8a5f81e41db5f796e7f64473454ec4fe2548e083c2fa0", size = 33387843, upload-time = "2026-07-02T06:37:05.097Z" }, + { url = "https://files.pythonhosted.org/packages/6d/b9/7708c43fed7ae28b4a1bad060b4221e3334cd827cec24f7165902a6ac1f4/av-18.0.0-cp311-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:ae56b40b6f8b067a8ad2dac664fbfbabac7f7a55b9a7bb031eb99289252bc017", size = 35536910, upload-time = "2026-07-02T06:37:08.806Z" }, + { url = "https://files.pythonhosted.org/packages/5a/94/eba99691d184f6a395a242d54dc370e2fd2265e95bbc98e2963a0fdbdd6c/av-18.0.0-cp311-abi3-manylinux_2_31_armv7l.whl", hash = "sha256:ea2e8ebbce521f21b55df9400e00d721623c9020ef158f5a188a96130be0743f", size = 38984619, upload-time = "2026-07-02T06:37:11.861Z" }, + { url = "https://files.pythonhosted.org/packages/c9/cf/0d7aee07fe16aa9ffdf96043c14bed5485a52c0dea4259de87aa306ecab4/av-18.0.0-cp311-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:ef96dabb3e50dac249913145dff5424b302b257fd95dcb64be3c7b7a8aef16d1", size = 34451176, upload-time = "2026-07-02T06:37:15.154Z" }, + { url = "https://files.pythonhosted.org/packages/76/92/810da80b12680d4c4fe235bd1b4003289be9213ac7f114b77b8ecf0e3b3e/av-18.0.0-cp311-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:0f65518a184613e41536f29e8758c8e3d8293e46bf5bef108f04f925bbfa3f44", size = 36619869, upload-time = "2026-07-02T06:37:18.495Z" }, + { url = "https://files.pythonhosted.org/packages/11/85/0f121ff43dc5a70696676c98a8f1674e2fa787614c2abaacb15fa1a9bc99/av-18.0.0-cp311-abi3-win_amd64.whl", hash = "sha256:aaf4d354d2beaa6651e4f92e54409a578bde64f79c0beef9a30b388d06f7c629", size = 27556236, upload-time = "2026-07-02T06:37:21.388Z" }, + { url = "https://files.pythonhosted.org/packages/8b/f6/2509754d4d2356abc6fc0ea3d57c12ade29bac23a1fb7fc215a53ca518fb/av-18.0.0-cp311-abi3-win_arm64.whl", hash = "sha256:adac2b3833b6cb9bd6cb52664a522b94db453615b3675b1dbb26e13fe1c80da6", size = 20221133, upload-time = "2026-07-02T06:37:23.88Z" }, +] + [[package]] name = "azure-cognitiveservices-speech" version = "1.48.2" @@ -738,6 +755,8 @@ dependencies = [ { name = "jaconv" }, { name = "jiwer" }, { name = "litellm" }, + { name = "livekit-agents" }, + { name = "livekit-plugins-elevenlabs" }, { name = "more-itertools" }, { name = "numpy" }, { name = "onnxruntime" }, @@ -797,6 +816,8 @@ requires-dist = [ { name = "jiwer", specifier = ">=3.0.0" }, { name = "librosa", marker = "extra == 'apps'", specifier = ">=0.11" }, { name = "litellm", specifier = "==1.85.0" }, + { name = "livekit-agents", specifier = ">=1.6.8" }, + { name = "livekit-plugins-elevenlabs", specifier = ">=1.6.8" }, { name = "more-itertools", specifier = ">=10.0.0" }, { name = "mypy", marker = "extra == 'dev'", specifier = ">=1.5" }, { name = "numpy", specifier = ">=1.24" }, @@ -829,6 +850,15 @@ requires-dist = [ ] provides-extras = ["apps", "dev"] +[[package]] +name = "eval-type-backport" +version = "0.4.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/1c/15/273a4baf8248d6d76220723c3caf039d283774b31a7c46ba686120145b76/eval_type_backport-0.4.0.tar.gz", hash = "sha256:8397d25e6524c2e67b9576bb0636be27dea2192017711220c534ec2de921e9b0", size = 10260, upload-time = "2026-06-02T13:22:06.059Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/50/a7/bb99bf5e6f78736ddb53480f2c3ff3702ffe2196a7c5e1661c03081d398e/eval_type_backport-0.4.0-py3-none-any.whl", hash = "sha256:ad5e2a8db71b6696a56eafb938b0f5a337d3217f256b8e158b469422b4772b20", size = 6432, upload-time = "2026-06-02T13:22:04.827Z" }, +] + [[package]] name = "fastapi" version = "0.133.1" @@ -1427,6 +1457,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7b/91/984aca2ec129e2757d1e4e3c81c3fcda9d0f85b74670a094cc443d9ee949/joblib-1.5.3-py3-none-any.whl", hash = "sha256:5fc3c5039fc5ca8c0276333a188bbd59d6b7ab37fe6632daa76bc7f9ec18e713", size = 309071, upload-time = "2025-12-15T08:41:44.973Z" }, ] +[[package]] +name = "json-repair" +version = "0.60.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/5e/a6/d69888cb4ffde30e80db1e6c32caaadd2f984a80067d5ea72c2cb3f61c3f/json_repair-0.60.1.tar.gz", hash = "sha256:841661cdd2df507c9a4e189097f38ca6bc372e06d4b4e36d72e590f68176c290", size = 49451, upload-time = "2026-06-03T17:28:44.451Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/32/1f/2a2b5eea8ef5762a86ad3f8fddddaaba2c0d76dd44e644b9158900868bec/json_repair-0.60.1-py3-none-any.whl", hash = "sha256:ba6ff974f2a8bef2f7768144a7f03f870a816443f03da27a49cdd0ec31a78049", size = 48045, upload-time = "2026-06-03T17:28:43.038Z" }, +] + [[package]] name = "jsonschema" version = "4.26.0" @@ -1562,6 +1601,157 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/1c/38/e6a4abb062e039d18d59538cc4e6fc370c2c10cd2bff4a2e546acb69dcb9/litellm-1.85.0-py3-none-any.whl", hash = "sha256:2bb449153610691faffd76f5b94a8c29e4b66fc5394156ebf54fd4fe92759b1a", size = 16978229, upload-time = "2026-05-17T01:59:11.902Z" }, ] +[[package]] +name = "livekit" +version = "1.1.14" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiofiles" }, + { name = "numpy" }, + { name = "protobuf" }, + { name = "types-protobuf" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ab/5d/bfaf1cc73f960b40294f604d334f05e628b0a07de3c47e475d760996a8d0/livekit-1.1.14.tar.gz", hash = "sha256:47428e10ecf20d7db4ee9fde4009bf96578c003b1ae6e1c5e7e4837a55902393", size = 375000, upload-time = "2026-07-31T14:05:14.425Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7a/ff/a2659522b3cf860b9b4453e1ec12d4b4c7e9cfd2b672f2cf925016d73492/livekit-1.1.14-py3-none-macosx_10_15_x86_64.whl", hash = "sha256:5f671b1752c93b878cb241b84fd3f72a31f857c3927755d672cfb7656a84778c", size = 10196322, upload-time = "2026-07-31T14:05:04.167Z" }, + { url = "https://files.pythonhosted.org/packages/82/a2/89f32d369cc78cb1a50b2a9e635c653f88d86ea4338ccdfa7b2d4ca0aecd/livekit-1.1.14-py3-none-macosx_11_0_arm64.whl", hash = "sha256:efa16b9036b0b592e5399fdb858c1f04ec8a32c385184c705f030952f72174e8", size = 9019745, upload-time = "2026-07-31T14:05:06.49Z" }, + { url = "https://files.pythonhosted.org/packages/8e/5b/dda7d660fa5d5b6e228dcfc6be3664a2442d1601481686052af2da642e5e/livekit-1.1.14-py3-none-manylinux_2_28_aarch64.whl", hash = "sha256:299146efefad5f67751cd15b8225bae759be0d7ad2f0b4ae1a22c15860d93cf9", size = 10042499, upload-time = "2026-07-31T14:05:08.563Z" }, + { url = "https://files.pythonhosted.org/packages/21/e3/d9255eeaf205f090d63d762e5254097b62af394bfaa90106f71f1fb6740e/livekit-1.1.14-py3-none-manylinux_2_28_x86_64.whl", hash = "sha256:80962c4a22ddbf0e0ebd3563fc090fce42df66b39b90de68b161b7db01970f68", size = 11445915, upload-time = "2026-07-31T14:05:10.628Z" }, + { url = "https://files.pythonhosted.org/packages/f7/0a/514fb230e7c7f13ae7e53b9e39a6dd9ea1aa9ff5be9e588d55301d159a1e/livekit-1.1.14-py3-none-win_amd64.whl", hash = "sha256:b8f8d38f131956297923e520bc4375bc9ebfa255cab7f125cb7755bfca71df24", size = 10766643, upload-time = "2026-07-31T14:05:12.716Z" }, +] + +[[package]] +name = "livekit-agents" +version = "1.6.8" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiofiles" }, + { name = "aiohttp" }, + { name = "av" }, + { name = "certifi" }, + { name = "click" }, + { name = "colorama" }, + { name = "docstring-parser" }, + { name = "eval-type-backport" }, + { name = "json-repair" }, + { name = "livekit" }, + { name = "livekit-api" }, + { name = "livekit-blingfire" }, + { name = "livekit-local-inference" }, + { name = "livekit-protocol" }, + { name = "nest-asyncio" }, + { name = "numpy" }, + { name = "openai" }, + { name = "opentelemetry-api" }, + { name = "opentelemetry-exporter-otlp" }, + { name = "opentelemetry-sdk" }, + { name = "prometheus-client" }, + { name = "protobuf" }, + { name = "psutil" }, + { name = "pydantic" }, + { name = "pyjwt" }, + { name = "pyyaml" }, + { name = "sounddevice" }, + { name = "typer" }, + { name = "types-protobuf" }, + { name = "typing-extensions" }, + { name = "watchfiles" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1a/71/168a1f61a23d652e72ea6f09bc3096f82a8b4f4d1d7f180dfdcffa6a1eba/livekit_agents-1.6.8.tar.gz", hash = "sha256:c25666b35ff44f19186cc5247690f4d0ec0737c312cc3e530af54abaea03ce7e", size = 2635925, upload-time = "2026-08-03T17:11:26.082Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/84/39/4ea6fea005f9c92cd444b04e517befcbf8f2fb8a9591fbfd5edfaa7a6a32/livekit_agents-1.6.8-py3-none-any.whl", hash = "sha256:fefe45142f398895f3fcb37f381569c2f9ce98b7eb0d9dfe6698aa9774d0f355", size = 2749031, upload-time = "2026-08-03T17:11:23.93Z" }, +] + +[package.optional-dependencies] +codecs = [ + { name = "numpy" }, +] + +[[package]] +name = "livekit-api" +version = "1.2.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiohttp" }, + { name = "livekit-protocol" }, + { name = "protobuf" }, + { name = "pyjwt" }, + { name = "types-protobuf" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/f3/19/36ff6712ec638a4b7dad4d8f03795952e401dc31db0b04cddec7892650da/livekit_api-1.2.0.tar.gz", hash = "sha256:a89817b3bca9584873786ff07209839308217537a42f95ecb2609aafaa109ddc", size = 20778, upload-time = "2026-07-11T23:20:54.781Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bf/e7/8926f16d4bc1b2e0ae46d4a507321bb899396d263a757f1adaabcd3b3867/livekit_api-1.2.0-py3-none-any.whl", hash = "sha256:307f8e5cfb0358c3ca091814ab768af55896022151bcd7f951954ccefa036a24", size = 26499, upload-time = "2026-07-11T23:20:53.736Z" }, +] + +[[package]] +name = "livekit-blingfire" +version = "1.1.0" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fc/09/1095ace608a41810d5c0f343eff36154505487c415acd9c653a882ff2cf1/livekit_blingfire-1.1.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:0358058ba6cba59379d22a01acef6ff8a729b0facf880c0f75d13c26f1315c9d", size = 153650, upload-time = "2025-12-16T00:48:05.976Z" }, + { url = "https://files.pythonhosted.org/packages/80/a5/f4eb0e5d97334581440d37ced2a1db4fdfc8454c641c7c144e858012f1ce/livekit_blingfire-1.1.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:a0741a8abcfaa1f3af2313271f15ac0f79777681a8e3ab9a782a68d8eb121c89", size = 148628, upload-time = "2025-12-16T00:48:06.998Z" }, + { url = "https://files.pythonhosted.org/packages/89/f9/dc5ad008cb8b9c2a300bb7f7d44f022cd4970a32707eb90358290a07f0e1/livekit_blingfire-1.1.0-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d99d7a34c9350da3a6ea738bc282a5f5b4ac4ffb7f8aa5251dfa96070ad845f6", size = 166832, upload-time = "2025-12-16T00:48:07.919Z" }, + { url = "https://files.pythonhosted.org/packages/8f/27/408c435cbed31fa3601ff32ef0499ff594cd898b483c9b4017e9df906de6/livekit_blingfire-1.1.0-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:815aca6c2f823fa25d7a15d8d76ce18b0295aa5ce2c988ed64fdbd9c4d3ced0a", size = 173959, upload-time = "2025-12-16T00:48:09.153Z" }, + { url = "https://files.pythonhosted.org/packages/2f/12/c826a40b32bfda29e7f826e50dfbd3c0a70726cb8c0cb5023d2311823bd2/livekit_blingfire-1.1.0-cp311-cp311-win_amd64.whl", hash = "sha256:7ae045d44d8cb867fc449f44a95c0287f6e5d225e62e24f4574bac8f26ede845", size = 130006, upload-time = "2025-12-16T00:48:10.176Z" }, + { url = "https://files.pythonhosted.org/packages/dd/18/8be31c84e911218011e6e653ca466fef320a4e7bc926aa694bc4cb6625f9/livekit_blingfire-1.1.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:9d5fb6746263529b780dc8bf7a6e6a80ff5fa7fa729e403f2b925996d041e039", size = 154567, upload-time = "2025-12-16T00:48:11.097Z" }, + { url = "https://files.pythonhosted.org/packages/03/64/bb5463d4a6a97888d52caa6256d242acab1f7eabcc59343f7874a89a30dc/livekit_blingfire-1.1.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:d4d5642e36fc0a9f89a5154affbd12305ae008c34c7b32f00fe00127ab18d6bd", size = 148792, upload-time = "2025-12-16T00:48:12.324Z" }, + { url = "https://files.pythonhosted.org/packages/6b/9f/ec51ebce455e17b6f304044e2bda57b15b1b45fd20b2feefa6e242fa33c6/livekit_blingfire-1.1.0-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:502d7a41fed246ec9cc432646d523c488a05fb2e572187a754735532ba5d69b7", size = 167606, upload-time = "2025-12-16T00:48:13.611Z" }, + { url = "https://files.pythonhosted.org/packages/d1/19/a4b56e54af456f2667287497f7678ff69a82ad21a687fc540213b4f25982/livekit_blingfire-1.1.0-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cdec36ea4d8b0dcda2791358ac9965e832539ecf13e651011197bab9960ea156", size = 174972, upload-time = "2025-12-16T00:48:14.811Z" }, + { url = "https://files.pythonhosted.org/packages/32/29/032cbf2c88ca40bee25b8a1b5346b5cb66487e689c4f42dd19f7e745090d/livekit_blingfire-1.1.0-cp312-cp312-win_amd64.whl", hash = "sha256:28d8c822616ca2ce53125040dfe09d06a6cc3e63c9055d39ca767a5c8f67ef84", size = 131026, upload-time = "2025-12-16T00:48:16.047Z" }, + { url = "https://files.pythonhosted.org/packages/81/50/46e410b935154a6bcf2d9494ee8e298b1a9c91ae33beaa78346703cf7681/livekit_blingfire-1.1.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:f5f6a40e498940f5b2e53d9753f5f7fb7f909e12a93a158844c9e3e99a5486b8", size = 154623, upload-time = "2025-12-16T00:48:17.641Z" }, + { url = "https://files.pythonhosted.org/packages/de/b4/f51c25bf104e51703dc66558ff9831a9769a9effa397956268902784a3d0/livekit_blingfire-1.1.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:945a672a224c9a686925e9af94c2660bacdbe190ccf693d6f17cea9359426c15", size = 148846, upload-time = "2025-12-16T00:48:18.591Z" }, + { url = "https://files.pythonhosted.org/packages/e0/d2/ad95d195ed6dccb6527ed3c1e753f211c3e9509050af5cddf007608bb104/livekit_blingfire-1.1.0-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:8f3aac3207cdd88c62323e0b07c33a69aac79c544122a2ddfbecc6c721ca760c", size = 167886, upload-time = "2025-12-16T00:48:19.858Z" }, + { url = "https://files.pythonhosted.org/packages/c5/67/fc4af1bbbed319d8edc319051bce720b51fa544f5d2ebb3201240779f135/livekit_blingfire-1.1.0-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:839feefa2910f99d794d3f3d696f95193ee8188cc6688a8d712bade2cede7951", size = 175858, upload-time = "2025-12-16T00:48:21.144Z" }, + { url = "https://files.pythonhosted.org/packages/76/6c/9e14763826476925767b511531318a83f95f3bf9e4dbc7dc611400af6e9e/livekit_blingfire-1.1.0-cp313-cp313-win_amd64.whl", hash = "sha256:1409d4c297260b60a37bfe6ba21e4fb59dd53cd929632c0a78a28d41fe424302", size = 131048, upload-time = "2025-12-16T00:48:22.17Z" }, +] + +[[package]] +name = "livekit-local-inference" +version = "0.2.6" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ca/1c/8010498321cb78111194dce2a1b400fd4548a6df9b88b469c3cf2787efd9/livekit_local_inference-0.2.6-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:d31c99cfcb486b7183381289e72f6f75594c44449f5f28e04fbdd202d4200f2c", size = 34833439, upload-time = "2026-06-24T16:54:48.636Z" }, + { url = "https://files.pythonhosted.org/packages/cc/c5/c371b7f1e36dfeab9d240a3e607320918c2564b51db7d2edb8f5064073c8/livekit_local_inference-0.2.6-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:56f0641306bc5451502f2f6597bac9b77e56487311d49a1ed356c5ed8ce01da6", size = 35099320, upload-time = "2026-06-24T16:54:51.761Z" }, + { url = "https://files.pythonhosted.org/packages/9a/04/914b672e43b619f0f6023641cd4ff539354ef07b12d79c2e0a538c219069/livekit_local_inference-0.2.6-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c684ee2d2f22c0a24ff471cdd5d873ee010135a9469d791c2cd2d3f83b219b51", size = 34821645, upload-time = "2026-06-24T16:54:57.782Z" }, + { url = "https://files.pythonhosted.org/packages/1d/a7/2d19b23b872a8e8b80efc612c3bfcc2f4733b24ca159c9d9426eff2d4eac/livekit_local_inference-0.2.6-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d6379f9d5ee4753919d10a2cedd2e16e1cbf634e496ad21fb54564513b731e69", size = 34865997, upload-time = "2026-06-24T16:55:01.294Z" }, + { url = "https://files.pythonhosted.org/packages/6f/66/250dbc92f4dd26b7c91c3f1c17ad238117339399006476728bfbabb5fb46/livekit_local_inference-0.2.6-cp311-cp311-win_amd64.whl", hash = "sha256:e5026b2cfd5aa2c85d677cce426b39bab88ee114acbaa814941d3aca8612ba7e", size = 34905736, upload-time = "2026-06-24T16:55:04.213Z" }, + { url = "https://files.pythonhosted.org/packages/a6/aa/7d6cfa6a2fe6baee8a443820d61b0e7df0cc7b9246762cab8f47c17ddcd3/livekit_local_inference-0.2.6-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:cb11919621cc542148ebed92408f0567b758057881c7e40d852d9738415a8eae", size = 34835713, upload-time = "2026-06-24T16:55:07.043Z" }, + { url = "https://files.pythonhosted.org/packages/be/57/e284fb2663bceb78f24496edf6d9468c861797fd24e655381c17fafe300f/livekit_local_inference-0.2.6-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:274ea4046e24377e0744056db051c53976d04bd159c12f8c482246d44089ae79", size = 35100531, upload-time = "2026-06-24T16:55:10.65Z" }, + { url = "https://files.pythonhosted.org/packages/76/6f/9a580a0d3f0c4bb63e7ab627467aeaf66980d21ea52e5946eb054e6f8150/livekit_local_inference-0.2.6-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:425469dff5505b35a3fab587993b320c829826d041a85c17c53475f64b99f444", size = 34823583, upload-time = "2026-06-24T16:55:13.728Z" }, + { url = "https://files.pythonhosted.org/packages/72/3d/c4a73813247faa704e9eef1cbf52ebc0f8070ad08b295419033ec3bcea5c/livekit_local_inference-0.2.6-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c0cc28d6e2e1431940b37a8c6d931e4dac914806e2c874a05513db6b65a01d7e", size = 34868325, upload-time = "2026-06-24T16:55:16.789Z" }, + { url = "https://files.pythonhosted.org/packages/c3/44/4f5a633f962142480a54eff1e4aea96848e4b54972c617da65e234d51b7d/livekit_local_inference-0.2.6-cp312-cp312-win_amd64.whl", hash = "sha256:5dc4a0574dc2e4a0745e8fcf29c26ca9f3983b885e63ada9761d26311a778d7b", size = 34906570, upload-time = "2026-06-24T16:55:19.914Z" }, + { url = "https://files.pythonhosted.org/packages/33/ae/5080f01d9c412a0702972efbfc8e4aca48f81baca494027aefcff5e4c8cc/livekit_local_inference-0.2.6-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:63beb204b64cdcbb3bc5b1caa1dbff2a9049998b6f897cf6b68285bb6b15926e", size = 34835783, upload-time = "2026-06-24T16:55:22.927Z" }, + { url = "https://files.pythonhosted.org/packages/25/17/53ee00d6abf7b9c7ad9036356c38f40634aa83f74b8484f40678a92264ad/livekit_local_inference-0.2.6-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:26e397fe543abac838c12101267e46974afd268b6475642fdf116391eee8bd43", size = 35100553, upload-time = "2026-06-24T16:55:26.351Z" }, + { url = "https://files.pythonhosted.org/packages/df/68/1fbbc4d22cbe28800380cdb1c51b8fadbf81bfacb0e966326fc7ee5c1298/livekit_local_inference-0.2.6-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:188457491cd59ed201d08ec3012fbc2a35d1fa5ac7bd4f4501553b16f5fc7d5b", size = 34823608, upload-time = "2026-06-24T16:55:29.4Z" }, + { url = "https://files.pythonhosted.org/packages/1a/7a/f62bc8dbdd6a4f32abe056bff1c4c4ad7ef30c780a43564594c72af120c2/livekit_local_inference-0.2.6-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:10e1866f23ff6ee694a360e8a032078de3df189d100a748295c30fd1013aef34", size = 34868352, upload-time = "2026-06-24T16:55:32.611Z" }, + { url = "https://files.pythonhosted.org/packages/c8/86/cd258845ad0b52e3ba68f776f3bc40eebf03d1dde31b2bdcd4d0a1fcb46a/livekit_local_inference-0.2.6-cp313-cp313-win_amd64.whl", hash = "sha256:69bc8976d8feef5a9c31e2c7bbd10d84e5494e1ecf85aedc75ebecef04fc9c08", size = 34906535, upload-time = "2026-06-24T16:55:35.78Z" }, +] + +[[package]] +name = "livekit-plugins-elevenlabs" +version = "1.6.8" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "livekit-agents", extra = ["codecs"] }, +] +sdist = { url = "https://files.pythonhosted.org/packages/fc/d7/83b43e8ba682e27eef01a14c949779aac1bfab435dd73d5d27157b541ba7/livekit_plugins_elevenlabs-1.6.8.tar.gz", hash = "sha256:a866a689ad7041f4576f3f2c0ce60db734213a8f4ba115cc995c6782a9f36af9", size = 19224, upload-time = "2026-08-03T17:11:52.752Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7d/d1/24d0696029f99bb1ce2776958ab06866962157bca07df04fec5209733897/livekit_plugins_elevenlabs-1.6.8-py3-none-any.whl", hash = "sha256:b82c5962d9aa2f80585c358ff2c7a1d2480309ce3914ae1d780d0a40fcbea2e7", size = 21942, upload-time = "2026-08-03T17:11:51.706Z" }, +] + +[[package]] +name = "livekit-protocol" +version = "1.1.22" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "protobuf" }, + { name = "types-protobuf" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/9b/65/736a378c2bf89c7fb54c1ff996f0bdc2046c588029d42afa2531b04c717d/livekit_protocol-1.1.22.tar.gz", hash = "sha256:a6517fd4ecea01ccd5055a30caefedc69a3e9ee02f715a79409b671091606692", size = 122570, upload-time = "2026-08-04T20:15:36.285Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/61/af/343dcfc7e429fbbc30993a733efb7fbfbbae70ac035a52e44111c6f4a76b/livekit_protocol-1.1.22-py3-none-any.whl", hash = "sha256:5c2edc843a48fe21d05b82c637c3e9eb92a88a34ba1e2c39857aca0e6105a84f", size = 149401, upload-time = "2026-08-04T20:15:35.073Z" }, +] + [[package]] name = "llvmlite" version = "0.44.0" @@ -1612,7 +1802,7 @@ name = "markdown-it-py" version = "4.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "mdurl", marker = "python_full_version >= '3.13'" }, + { name = "mdurl" }, ] sdist = { url = "https://files.pythonhosted.org/packages/5b/f5/4ec618ed16cc4f8fb3b701563655a69816155e79e24a17b651541804721d/markdown_it_py-4.0.0.tar.gz", hash = "sha256:cb0a2b4aa34f932c007117b194e945bd74e0ec24133ceb5bac59009cda1cb9f3", size = 73070, upload-time = "2025-08-11T12:57:52.854Z" } wheels = [ @@ -1865,6 +2055,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/4b/27/20770bd6bf8fbe1e16f848ba21da9df061f38d2e6483952c29d2bb5d1d8b/narwhals-2.17.0-py3-none-any.whl", hash = "sha256:2ac5307b7c2b275a7d66eeda906b8605e3d7a760951e188dcfff86e8ebe083dd", size = 444897, upload-time = "2026-02-23T09:44:32.006Z" }, ] +[[package]] +name = "nest-asyncio" +version = "1.6.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/83/f8/51569ac65d696c8ecbee95938f89d4abf00f47d58d48f6fbabfe8f0baefe/nest_asyncio-1.6.0.tar.gz", hash = "sha256:6f172d5449aca15afd6c646851f4e31e02c598d553a667e38cafa997cfec55fe", size = 7418, upload-time = "2024-01-21T14:25:19.227Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a0/c4/c2971a3ba4c6103a3d10c4b0f24f461ddc027f0f09763220cf35ca1401b3/nest_asyncio-1.6.0-py3-none-any.whl", hash = "sha256:87af6efd6b5e897c81050477ef65c62e2b2f35d51703cae01aff2905b1852e1c", size = 5195, upload-time = "2024-01-21T14:25:17.223Z" }, +] + [[package]] name = "nltk" version = "3.9.4" @@ -2014,6 +2213,118 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9d/1c/5d43735b2553baae2a5e899dcbcd0670a86930d993184d72ca909bf11c9b/openai-2.36.0-py3-none-any.whl", hash = "sha256:143f6194b548dbc2c921af1f1b03b9f14c85fed8a75b5b516f5bcc11a2a50c63", size = 1302361, upload-time = "2026-05-07T17:33:15.063Z" }, ] +[[package]] +name = "opentelemetry-api" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ee/8b/aa9e2d8b8dfa7c946f7dec5d1f8f6ba8eca062f43509a06bdb5ce93d26c0/opentelemetry_api-1.44.0.tar.gz", hash = "sha256:67647e5e9566edcf421166fdf022b3537f818635daa852b289e34604dc6fb33a", size = 72406, upload-time = "2026-07-16T15:25:32.678Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ca/6f/a04e900f465ff3221ccc395522503e2d10e79fa21f2723c8e177aae1e0d1/opentelemetry_api-1.44.0-py3-none-any.whl", hash = "sha256:94b98c893a91b88657eaac1e3ba89618cdb85be6918196705354f34728b2cdef", size = 60018, upload-time = "2026-07-16T15:25:11.657Z" }, +] + +[[package]] +name = "opentelemetry-exporter-otlp" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-exporter-otlp-proto-grpc" }, + { name = "opentelemetry-exporter-otlp-proto-http" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/f2/45/7af37fe54e5d3e66e7dcd7ba8b8aeee73f202bfac909cc94b8c4e428f9ac/opentelemetry_exporter_otlp-1.44.0.tar.gz", hash = "sha256:af1cde7c33ea8ed624bf04ac49a885730fe44c1f1ad698656e592c38f70ce106", size = 6090, upload-time = "2026-07-16T15:25:34.585Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/18/c3/7b466a9463944e70b37b744072a0c1b88a425dade3fff0631adec66c9bcc/opentelemetry_exporter_otlp-1.44.0-py3-none-any.whl", hash = "sha256:4a498fa8d8fd8be9e8e2d175fe5524a3fe581ccffadd8509db86526a5fb97051", size = 6727, upload-time = "2026-07-16T15:25:14.445Z" }, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-common" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-proto" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/61/09/4d717852c1cf3f854b76c7110a5d00883bc3c99288b9b0dbcbeb9e306eb6/opentelemetry_exporter_otlp_proto_common-1.44.0.tar.gz", hash = "sha256:dc87a5a5bc58f149a56d1547e4691588fa12994cdc3bc039a694ccb3375862ac", size = 20202, upload-time = "2026-07-16T15:25:37.658Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5e/71/65fd9d54c10b860f87c045ccee1264cab7011268895d3528818a29c1172a/opentelemetry_exporter_otlp_proto_common-1.44.0-py3-none-any.whl", hash = "sha256:9a9fe61bba73d802904bc989f1d6b4a7b1ee40f06c40e98d6f85af65aaebb694", size = 17045, upload-time = "2026-07-16T15:25:18.201Z" }, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-grpc" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "googleapis-common-protos" }, + { name = "grpcio" }, + { name = "opentelemetry-api" }, + { name = "opentelemetry-exporter-otlp-proto-common" }, + { name = "opentelemetry-proto" }, + { name = "opentelemetry-sdk" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1f/47/80d9e9d468dc5de3af5096f5ccdb065fa4dd1470f74495cc53e59e397f47/opentelemetry_exporter_otlp_proto_grpc-1.44.0.tar.gz", hash = "sha256:40d1ae9e03fcc36de3cbac610cc99f35894938bff9cfd90fc4ec68bd85448463", size = 27225, upload-time = "2026-07-16T15:25:38.308Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/54/29/6ae42ba32b153ae0a44ae125f0caff2188bbe62d99c82d1768da30864e72/opentelemetry_exporter_otlp_proto_grpc-1.44.0-py3-none-any.whl", hash = "sha256:6a1a645ea182a2f59440c51fa8301d309f3324a8f9d65f8395584b064b67ee4e", size = 19624, upload-time = "2026-07-16T15:25:19.096Z" }, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-http" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "googleapis-common-protos" }, + { name = "opentelemetry-api" }, + { name = "opentelemetry-exporter-otlp-proto-common" }, + { name = "opentelemetry-proto" }, + { name = "opentelemetry-sdk" }, + { name = "requests" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1a/87/95e2a5aaa795b4e2260d74e16df2d5541deb2ea9de010bcd615f4dee2654/opentelemetry_exporter_otlp_proto_http-1.44.0.tar.gz", hash = "sha256:c633d7270ad6b57cd4cfbe8b0007a9e2e7c0cb50bd6c50fe2a7b245f721a09d8", size = 25806, upload-time = "2026-07-16T15:25:39.162Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/cd/d0/fdeb1a98d8d3a6205f5f297c51b4a9bfe65126ab60339669bbe3dd54c2e2/opentelemetry_exporter_otlp_proto_http-1.44.0-py3-none-any.whl", hash = "sha256:838592fce774c1c8bb7b9a0a7facbfa82e17be5a8a4e94cef10cb84ae026bae3", size = 21850, upload-time = "2026-07-16T15:25:20.006Z" }, +] + +[[package]] +name = "opentelemetry-proto" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "protobuf" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/64/01/40ac4ae9a149263cc52c2cee200ddd80cb6d8db1a4610abf8eabce0fe771/opentelemetry_proto-1.44.0.tar.gz", hash = "sha256:c547a79c2f8c0c515d31509154682e5921c7cfd5ca67b70e1f9266e2c3e103f3", size = 46488, upload-time = "2026-07-16T15:25:45.34Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d1/7c/8be563d68e93bbefa5c8affb82ddcff91b3ad858ce49957ba7b16fd3e0ab/opentelemetry_proto-1.44.0-py3-none-any.whl", hash = "sha256:898b155a0e1557afd867478fb6158e8122a46329ca0bb8dc53cc55e98f017f56", size = 72483, upload-time = "2026-07-16T15:25:28.429Z" }, +] + +[[package]] +name = "opentelemetry-sdk" +version = "1.44.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "opentelemetry-semantic-conventions" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/5d/77/a6592cbc7c8d9bcc9d6757a9df45e04a7c585e3e6e7a13456da522b21109/opentelemetry_sdk-1.44.0.tar.gz", hash = "sha256:cebe7f65dc12f26ead75c6064de12fd2a9052e5060c0272d402cfa203aae123b", size = 208624, upload-time = "2026-07-16T15:25:46.078Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e7/23/ff077e61886ee020a17ce9c8b6fa11c601c8d8345b09ea24f605445df62a/opentelemetry_sdk-1.44.0-py3-none-any.whl", hash = "sha256:df081c4c6bcfdb1211e3e86140376792643128a25f8d72d1d27675936e7e96ad", size = 137221, upload-time = "2026-07-16T15:25:29.534Z" }, +] + +[[package]] +name = "opentelemetry-semantic-conventions" +version = "0.65b0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "opentelemetry-api" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/8f/73/0cbdebcb4cf545fdd328da14f5137e37d0770c3f26185e478b0d15d94f50/opentelemetry_semantic_conventions-0.65b0.tar.gz", hash = "sha256:f9b2b81e9d5b64f11bc952075e7e9c7fb0aab075c7fd1c46d597f1b919852d60", size = 148774, upload-time = "2026-07-16T15:25:46.902Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a6/0e/49df70d9b81fb5cbae4bbf2a49d865b09bcbcbc4eb53f5851b1027738d78/opentelemetry_semantic_conventions-0.65b0-py3-none-any.whl", hash = "sha256:1cacde7b0ad306f84c5ef08c3dbe1bbaf20165bba6f8bff43b670e555a086bcb", size = 204645, upload-time = "2026-07-16T15:25:30.688Z" }, +] + [[package]] name = "packaging" version = "26.0" @@ -2241,6 +2552,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/5d/19/fd3ef348460c80af7bb4669ea7926651d1f95c23ff2df18b9d24bab4f3fa/pre_commit-4.5.1-py2.py3-none-any.whl", hash = "sha256:3b3afd891e97337708c1674210f8eba659b52a38ea5f822ff142d10786221f77", size = 226437, upload-time = "2025-12-16T21:14:32.409Z" }, ] +[[package]] +name = "prometheus-client" +version = "0.26.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/52/73/f1334c29c2af4cd9dba6c7817e61b611bd0215e2eb5565c6064a4de18802/prometheus_client-0.26.0.tar.gz", hash = "sha256:04a91bcf94e2cf74a44a1a874d651a2e853ed354b6e822f3b7487751465d5c2b", size = 92910, upload-time = "2026-07-24T19:36:41.893Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/eb/a3/b69efbf4143b5b9859b977770bbbabcc2796b702fa69dc40271e45cd5a56/prometheus_client-0.26.0-py3-none-any.whl", hash = "sha256:fa93d06737aa02bacd05794768508bb97d2fbee28cb3bca04eaae92f0ca953d6", size = 64494, upload-time = "2026-07-24T19:36:40.854Z" }, +] + [[package]] name = "propcache" version = "0.4.1" @@ -2337,6 +2657,28 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/c4/72/02445137af02769918a93807b2b7890047c32bfb9f90371cbc12688819eb/protobuf-6.33.6-py3-none-any.whl", hash = "sha256:77179e006c476e69bf8e8ce866640091ec42e1beb80b213c3900006ecfba6901", size = 170656, upload-time = "2026-03-18T19:04:59.826Z" }, ] +[[package]] +name = "psutil" +version = "7.2.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/aa/c6/d1ddf4abb55e93cebc4f2ed8b5d6dbad109ecb8d63748dd2b20ab5e57ebe/psutil-7.2.2.tar.gz", hash = "sha256:0746f5f8d406af344fd547f1c8daa5f5c33dbc293bb8d6a16d80b4bb88f59372", size = 493740, upload-time = "2026-01-28T18:14:54.428Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/51/08/510cbdb69c25a96f4ae523f733cdc963ae654904e8db864c07585ef99875/psutil-7.2.2-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:2edccc433cbfa046b980b0df0171cd25bcaeb3a68fe9022db0979e7aa74a826b", size = 130595, upload-time = "2026-01-28T18:14:57.293Z" }, + { url = "https://files.pythonhosted.org/packages/d6/f5/97baea3fe7a5a9af7436301f85490905379b1c6f2dd51fe3ecf24b4c5fbf/psutil-7.2.2-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:e78c8603dcd9a04c7364f1a3e670cea95d51ee865e4efb3556a3a63adef958ea", size = 131082, upload-time = "2026-01-28T18:14:59.732Z" }, + { url = "https://files.pythonhosted.org/packages/37/d6/246513fbf9fa174af531f28412297dd05241d97a75911ac8febefa1a53c6/psutil-7.2.2-cp313-cp313t-manylinux2010_x86_64.manylinux_2_12_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1a571f2330c966c62aeda00dd24620425d4b0cc86881c89861fbc04549e5dc63", size = 181476, upload-time = "2026-01-28T18:15:01.884Z" }, + { url = "https://files.pythonhosted.org/packages/b8/b5/9182c9af3836cca61696dabe4fd1304e17bc56cb62f17439e1154f225dd3/psutil-7.2.2-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:917e891983ca3c1887b4ef36447b1e0873e70c933afc831c6b6da078ba474312", size = 184062, upload-time = "2026-01-28T18:15:04.436Z" }, + { url = "https://files.pythonhosted.org/packages/16/ba/0756dca669f5a9300d0cbcbfae9a4c30e446dfc7440ffe43ded5724bfd93/psutil-7.2.2-cp313-cp313t-win_amd64.whl", hash = "sha256:ab486563df44c17f5173621c7b198955bd6b613fb87c71c161f827d3fb149a9b", size = 139893, upload-time = "2026-01-28T18:15:06.378Z" }, + { url = "https://files.pythonhosted.org/packages/1c/61/8fa0e26f33623b49949346de05ec1ddaad02ed8ba64af45f40a147dbfa97/psutil-7.2.2-cp313-cp313t-win_arm64.whl", hash = "sha256:ae0aefdd8796a7737eccea863f80f81e468a1e4cf14d926bd9b6f5f2d5f90ca9", size = 135589, upload-time = "2026-01-28T18:15:08.03Z" }, + { url = "https://files.pythonhosted.org/packages/e7/36/5ee6e05c9bd427237b11b3937ad82bb8ad2752d72c6969314590dd0c2f6e/psutil-7.2.2-cp36-abi3-macosx_10_9_x86_64.whl", hash = "sha256:ed0cace939114f62738d808fdcecd4c869222507e266e574799e9c0faa17d486", size = 129090, upload-time = "2026-01-28T18:15:22.168Z" }, + { url = "https://files.pythonhosted.org/packages/80/c4/f5af4c1ca8c1eeb2e92ccca14ce8effdeec651d5ab6053c589b074eda6e1/psutil-7.2.2-cp36-abi3-macosx_11_0_arm64.whl", hash = "sha256:1a7b04c10f32cc88ab39cbf606e117fd74721c831c98a27dc04578deb0c16979", size = 129859, upload-time = "2026-01-28T18:15:23.795Z" }, + { url = "https://files.pythonhosted.org/packages/b5/70/5d8df3b09e25bce090399cf48e452d25c935ab72dad19406c77f4e828045/psutil-7.2.2-cp36-abi3-manylinux2010_x86_64.manylinux_2_12_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:076a2d2f923fd4821644f5ba89f059523da90dc9014e85f8e45a5774ca5bc6f9", size = 155560, upload-time = "2026-01-28T18:15:25.976Z" }, + { url = "https://files.pythonhosted.org/packages/63/65/37648c0c158dc222aba51c089eb3bdfa238e621674dc42d48706e639204f/psutil-7.2.2-cp36-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b0726cecd84f9474419d67252add4ac0cd9811b04d61123054b9fb6f57df6e9e", size = 156997, upload-time = "2026-01-28T18:15:27.794Z" }, + { url = "https://files.pythonhosted.org/packages/8e/13/125093eadae863ce03c6ffdbae9929430d116a246ef69866dad94da3bfbc/psutil-7.2.2-cp36-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:fd04ef36b4a6d599bbdb225dd1d3f51e00105f6d48a28f006da7f9822f2606d8", size = 148972, upload-time = "2026-01-28T18:15:29.342Z" }, + { url = "https://files.pythonhosted.org/packages/04/78/0acd37ca84ce3ddffaa92ef0f571e073faa6d8ff1f0559ab1272188ea2be/psutil-7.2.2-cp36-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b58fabe35e80b264a4e3bb23e6b96f9e45a3df7fb7eed419ac0e5947c61e47cc", size = 148266, upload-time = "2026-01-28T18:15:31.597Z" }, + { url = "https://files.pythonhosted.org/packages/b4/90/e2159492b5426be0c1fef7acba807a03511f97c5f86b3caeda6ad92351a7/psutil-7.2.2-cp37-abi3-win_amd64.whl", hash = "sha256:eb7e81434c8d223ec4a219b5fc1c47d0417b12be7ea866e24fb5ad6e84b3d988", size = 137737, upload-time = "2026-01-28T18:15:33.849Z" }, + { url = "https://files.pythonhosted.org/packages/8c/c7/7bb2e321574b10df20cbde462a94e2b71d05f9bbda251ef27d104668306a/psutil-7.2.2-cp37-abi3-win_arm64.whl", hash = "sha256:8c233660f575a5a89e6d4cb65d9f938126312bca76d8fe087b947b3a1aaac9ee", size = 134617, upload-time = "2026-01-28T18:15:36.514Z" }, +] + [[package]] name = "pyarrow" version = "23.0.1" @@ -2532,6 +2874,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/c7/21/705964c7812476f378728bdf590ca4b771ec72385c533964653c68e86bdc/pygments-2.19.2-py3-none-any.whl", hash = "sha256:86540386c03d588bb81d44bc3928634ff26449851e99741617ecb9037ee5ec0b", size = 1225217, upload-time = "2025-06-21T13:39:07.939Z" }, ] +[[package]] +name = "pyjwt" +version = "2.13.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/3b/81/58d0ac84e1ef3a3843791d6954d94c0b33d526c75eeb1efbce9d0a4c4077/pyjwt-2.13.0.tar.gz", hash = "sha256:41571c89ca91598c79e8ef18a2d07367d4810fbbd6f637794879baf1b7703423", size = 107515, upload-time = "2026-05-21T19:54:36.618Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a3/5e/ecf12fdb62546d64385c158514e9b2b671f7832108ef2ecd2020ce0af2d1/pyjwt-2.13.0-py3-none-any.whl", hash = "sha256:66adcc2aff09b3f1bbd95fc1e1577df8ac8723c978552fd43304c8a290ac5728", size = 31274, upload-time = "2026-05-21T19:54:35.362Z" }, +] + [[package]] name = "pyloudnorm" version = "0.2.0" @@ -2856,8 +3207,8 @@ name = "rich" version = "14.3.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "markdown-it-py", marker = "python_full_version >= '3.13'" }, - { name = "pygments", marker = "python_full_version >= '3.13'" }, + { name = "markdown-it-py" }, + { name = "pygments" }, ] sdist = { url = "https://files.pythonhosted.org/packages/b3/c6/f3b320c27991c46f43ee9d856302c70dc2d0fb2dba4842ff739d5f46b393/rich-14.3.3.tar.gz", hash = "sha256:b8daa0b9e4eef54dd8cf7c86c03713f53241884e814f4e2f5fb342fe520f639b", size = 230582, upload-time = "2026-02-19T17:23:12.474Z" } wheels = [ @@ -3114,6 +3465,22 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2", size = 10235, upload-time = "2024-02-25T23:20:01.196Z" }, ] +[[package]] +name = "sounddevice" +version = "0.5.5" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "cffi" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/2a/f9/2592608737553638fca98e21e54bfec40bf577bb98a61b2770c912aab25e/sounddevice-0.5.5.tar.gz", hash = "sha256:22487b65198cb5bf2208755105b524f78ad173e5ab6b445bdab1c989f6698df3", size = 143191, upload-time = "2026-01-23T18:36:43.529Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1e/0a/478e441fd049002cf308520c0d62dd8333e7c6cc8d997f0dda07b9fbcc46/sounddevice-0.5.5-py3-none-any.whl", hash = "sha256:30ff99f6c107f49d25ad16a45cacd8d91c25a1bcdd3e81a206b921a3a6405b1f", size = 32807, upload-time = "2026-01-23T18:36:35.649Z" }, + { url = "https://files.pythonhosted.org/packages/56/f9/c037c35f6d0b6bc3bc7bfb314f1d6f1f9a341328ef47cd63fc4f850a7b27/sounddevice-0.5.5-py3-none-macosx_10_6_x86_64.macosx_10_6_universal2.whl", hash = "sha256:05eb9fd6c54c38d67741441c19164c0dae8ce80453af2d8c4ad2e7823d15b722", size = 108557, upload-time = "2026-01-23T18:36:37.41Z" }, + { url = "https://files.pythonhosted.org/packages/88/a1/d19dd9889cd4bce2e233c4fac007cd8daaf5b9fe6e6a5d432cf17be0b807/sounddevice-0.5.5-py3-none-win32.whl", hash = "sha256:1234cc9b4c9df97b6cbe748146ae0ec64dd7d6e44739e8e42eaa5b595313a103", size = 317765, upload-time = "2026-01-23T18:36:39.047Z" }, + { url = "https://files.pythonhosted.org/packages/c3/0e/002ed7c4c1c2ab69031f78989d3b789fee3a7fba9e586eb2b81688bf4961/sounddevice-0.5.5-py3-none-win_amd64.whl", hash = "sha256:cfc6b2c49fb7f555591c78cb8ecf48d6a637fd5b6e1db5fec6ed9365d64b3519", size = 365324, upload-time = "2026-01-23T18:36:40.496Z" }, + { url = "https://files.pythonhosted.org/packages/4e/39/a61d4b83a7746b70d23d9173be688c0c6bfc7173772344b7442c2c155497/sounddevice-0.5.5-py3-none-win_arm64.whl", hash = "sha256:3861901ddd8230d2e0e8ae62ac320cdd4c688d81df89da036dcb812f757bb3e6", size = 317115, upload-time = "2026-01-23T18:36:42.235Z" }, +] + [[package]] name = "soundfile" version = "0.13.1" @@ -3462,16 +3829,25 @@ name = "typer" version = "0.24.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "annotated-doc", marker = "python_full_version >= '3.13'" }, - { name = "click", marker = "python_full_version >= '3.13'" }, - { name = "rich", marker = "python_full_version >= '3.13'" }, - { name = "shellingham", marker = "python_full_version >= '3.13'" }, + { name = "annotated-doc" }, + { name = "click" }, + { name = "rich" }, + { name = "shellingham" }, ] sdist = { url = "https://files.pythonhosted.org/packages/f5/24/cb09efec5cc954f7f9b930bf8279447d24618bb6758d4f6adf2574c41780/typer-0.24.1.tar.gz", hash = "sha256:e39b4732d65fbdcde189ae76cf7cd48aeae72919dea1fdfc16593be016256b45", size = 118613, upload-time = "2026-02-21T16:54:40.609Z" } wheels = [ { url = "https://files.pythonhosted.org/packages/4a/91/48db081e7a63bb37284f9fbcefda7c44c277b18b0e13fbc36ea2335b71e6/typer-0.24.1-py3-none-any.whl", hash = "sha256:112c1f0ce578bfb4cab9ffdabc68f031416ebcc216536611ba21f04e9aa84c9e", size = 56085, upload-time = "2026-02-21T16:54:41.616Z" }, ] +[[package]] +name = "types-protobuf" +version = "7.34.1.20260518" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/29/59/e2b13b499d15e6720150c4b1a8d91e31fcacf716b432397475b3151ff7e4/types_protobuf-7.34.1.20260518.tar.gz", hash = "sha256:28cfaded25889cb83ebfb63cfb0a43628f0b6f3785767bec17287dc6468795f2", size = 68936, upload-time = "2026-05-18T06:01:47.332Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2a/1f/ec5caf72c2e3b688ca3927e0979a04ddad19e1afc4bf1c199bd743e0f419/types_protobuf-7.34.1.20260518-py3-none-any.whl", hash = "sha256:a0a5337413347166439c0e07cbc26c6164d091401c6f01b1dfd8cdb966c4dd8f", size = 85992, upload-time = "2026-05-18T06:01:45.696Z" }, +] + [[package]] name = "typing-extensions" version = "4.15.0" @@ -3566,6 +3942,74 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/33/e8/e40370e6d74ddba47f002a32919d91310d6074130fe4e17dabcafc15cbf1/watchdog-6.0.0-py3-none-win_ia64.whl", hash = "sha256:a1914259fa9e1454315171103c6a30961236f508b9b623eae470268bbcc6a22f", size = 79067, upload-time = "2024-11-01T14:07:11.845Z" }, ] +[[package]] +name = "watchfiles" +version = "1.2.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/cd/41/5e1a4bb12aac5f1493fa1bdc11154eca3b258ca4eba65d39c473fe19d8e9/watchfiles-1.2.0.tar.gz", hash = "sha256:c995fba777f1ea992f090f9236e9284cf7a5d1a0130dd5a3d82c598cacd76838", size = 108252, upload-time = "2026-05-18T04:32:04.251Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fc/3d/8024c801df84d1587740d0359e7fdd80afeae3d159011f3d5376dd82f18e/watchfiles-1.2.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:704fd259e332e01f9b9c178f4bce9e49027e5587cc2600eeeaf8e76e1c846201", size = 400242, upload-time = "2026-05-18T04:31:19.014Z" }, + { url = "https://files.pythonhosted.org/packages/87/5b/f4dfd45323e949984a3a7f9dc31d1cbb049921e7d98253488dda72ccdaa9/watchfiles-1.2.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:6543cf55d170003296d185c0af981f3e1311564907e1f4e08671fc7693a890a5", size = 394562, upload-time = "2026-05-18T04:30:08.46Z" }, + { url = "https://files.pythonhosted.org/packages/98/d8/19483ef075d601c409bce8bcbb5c0f81a10876fff870400568f08ce484a1/watchfiles-1.2.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:89d8c2394a065ca86f5d2910ff263ae67c127e1376ccc4f9fc35c71db879f80a", size = 456611, upload-time = "2026-05-18T04:30:45.723Z" }, + { url = "https://files.pythonhosted.org/packages/b1/6a/cc81fbe7ee42f2f22e661a6e12def7807e01b14b2f39e0ff83fd373fd307/watchfiles-1.2.0-cp311-cp311-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:772b80df316480d894a0e3165fdd19cf77f5d17f9a787f94029465ad0e3529d1", size = 461379, upload-time = "2026-05-18T04:31:29.292Z" }, + { url = "https://files.pythonhosted.org/packages/b1/57/7e669002082c0a0f4fb5113bb70125f7110124b846b0a11bc5ae8e90eac1/watchfiles-1.2.0-cp311-cp311-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:d158cd89df6053823533e06fb1d73c549133bff5f0396170c0e53d9559340717", size = 493556, upload-time = "2026-05-18T04:30:05.44Z" }, + { url = "https://files.pythonhosted.org/packages/45/7d/f60a2b19807b21fe8281f3a8da4f59eef0d5f96825ac4680ba2d4f2ebf91/watchfiles-1.2.0-cp311-cp311-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:d516b3283a758e087841aedb8031549fb41ced08f3db10aa6d2bf32dc042525b", size = 575255, upload-time = "2026-05-18T04:30:40.568Z" }, + { url = "https://files.pythonhosted.org/packages/bd/49/77f5b5e6efbcd57482f74948ebb1b97e5c0046d6b61475042d830c84b3ff/watchfiles-1.2.0-cp311-cp311-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:53b2290c92e0506d102cd448fbc610d87079553f86caa39d67440856a8b8bba5", size = 467052, upload-time = "2026-05-18T04:31:17.942Z" }, + { url = "https://files.pythonhosted.org/packages/ee/5a/73e2959af1b97fd5d556f9a8bdba017be23ceeef731869d5eaa0a753d5a3/watchfiles-1.2.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a711b51aec4370d0dcda5b6c09463206f133a5759341d7744b953a7b62e1100e", size = 456858, upload-time = "2026-05-18T04:30:30.182Z" }, + { url = "https://files.pythonhosted.org/packages/50/57/1bc8c27fad7e6c19bddee15d276dbb6ab72480ec01c127afff1673aee417/watchfiles-1.2.0-cp311-cp311-manylinux_2_31_riscv64.whl", hash = "sha256:e2ca07fa7d89195ec0865d3d285666286740bfa83d83e5cee204043a31ecc165", size = 467579, upload-time = "2026-05-18T04:32:15.897Z" }, + { url = "https://files.pythonhosted.org/packages/09/6c/3c2e44edba3553c5e3c3b8c8a2a6dee6b9e12ae2cf4bd2378bebf9dc3038/watchfiles-1.2.0-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:e0618518f282c4ebff60f5e5b1247b6d91bb8b9f4476947563a1e74acc66f3c6", size = 633253, upload-time = "2026-05-18T04:31:37.123Z" }, + { url = "https://files.pythonhosted.org/packages/30/c2/d8c84a882ab39bbefcc4915ab3e91830b7a7e990c5570b0b69075aba3faf/watchfiles-1.2.0-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:0d191c054d0715c3c95c99df9b8dbf6fd096d8c1e021e8f212e1bd8bc444ccb5", size = 660713, upload-time = "2026-05-18T04:31:24.62Z" }, + { url = "https://files.pythonhosted.org/packages/a9/07/f97736a5fc605364fe67b25e9fa4a6965dfd4840d50c406ada507e9d735f/watchfiles-1.2.0-cp311-cp311-win32.whl", hash = "sha256:9342472aff9b093c5acd4f6d8f70ae0937964ab56542502bcf5579782da69ae8", size = 277222, upload-time = "2026-05-18T04:31:21.131Z" }, + { url = "https://files.pythonhosted.org/packages/cf/99/2b04981977fc2608afd60360d928c6aecf6b950292ca221d98f4005f6694/watchfiles-1.2.0-cp311-cp311-win_amd64.whl", hash = "sha256:dbd6c97045dad81227c8d040173da044c1de08de64a5ea8b555da4aee1d5fa22", size = 290274, upload-time = "2026-05-18T04:31:45.966Z" }, + { url = "https://files.pythonhosted.org/packages/3c/74/f7f58a7075ee9cf612b0cfcddb78b8cd8234f0742d6f0075cf0da2dde1c6/watchfiles-1.2.0-cp311-cp311-win_arm64.whl", hash = "sha256:57a2d9fa4fb4c2ecae57b13dfff2c7ab53e21a2ba674fe9f05506680fcdcc0d7", size = 283460, upload-time = "2026-05-18T04:31:39.126Z" }, + { url = "https://files.pythonhosted.org/packages/b8/2f/e42c992d2afda3108ea1c02acecc991b9f31d05c14adc2a7cee9ee211fc4/watchfiles-1.2.0-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:bc13eb17538be00c874699dc0abe4ee2bc8d50bb1166a6b9e175ef3fd7eb8f26", size = 400115, upload-time = "2026-05-18T04:32:02.06Z" }, + { url = "https://files.pythonhosted.org/packages/5f/8f/6af2ea19065c91d8b0ea3516fdfc8c0d349f407e8e9fbf4e5a17360de8ad/watchfiles-1.2.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:2d95ddc1eb6914154253d239089900813f6a767e174b8e6a50e7fdacb7e4236c", size = 393659, upload-time = "2026-05-18T04:30:50.951Z" }, + { url = "https://files.pythonhosted.org/packages/13/01/b32a967c56fb3e3e5be3db52c3d3b87fa4513aa367d8ed1ad96d42952e5f/watchfiles-1.2.0-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:8f70d8b291ef6e88d19b1f297a6905ddb978888d9272b0d05e6f53309856bcfc", size = 453207, upload-time = "2026-05-18T04:31:04.231Z" }, + { url = "https://files.pythonhosted.org/packages/04/98/97557a812180338cb1abd32e1cffcc4588f59b5f23e0cb006b2ba95ba64a/watchfiles-1.2.0-cp312-cp312-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:56d8641cf834c2836922899105bd3ce3d0dfc69291d52edf0b4d0436829b34c0", size = 459273, upload-time = "2026-05-18T04:31:50.377Z" }, + { url = "https://files.pythonhosted.org/packages/e8/a8/b4b08dcb7653b8087c6586f7ce649505900e866bbcfe40dc9587af02e686/watchfiles-1.2.0-cp312-cp312-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:2581a94056e55d7d0a31a823ea92bf73749c489ca2285bfdc0fbe6b2bb49d50c", size = 489927, upload-time = "2026-05-18T04:31:42.485Z" }, + { url = "https://files.pythonhosted.org/packages/50/94/3dceea03545d2e5ddfd839f0ddd5e1cecbf1697b5a428d5ba11cef6af95d/watchfiles-1.2.0-cp312-cp312-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:41bc1199f7523b3f82843c88cbb979180c949caef0342cf90968f178e5d49b01", size = 570476, upload-time = "2026-05-18T04:31:03.071Z" }, + { url = "https://files.pythonhosted.org/packages/cc/f2/d39a5450c3532092b91f81d274360e613c2371bc874a89c7a1a3c5e8d138/watchfiles-1.2.0-cp312-cp312-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:7571e4464cb6e434958f867f7f730b8ab0b75e3f8e5eac0499168486ab3c33a8", size = 465650, upload-time = "2026-05-18T04:30:12.701Z" }, + { url = "https://files.pythonhosted.org/packages/22/24/ed72f68cbc1333ca9b9f2200aa048bb6658ae41709bc1caad4310f4bdffd/watchfiles-1.2.0-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e53a384f76b631c3ae5334ce6a52f0baa3a911eb94a4eac7f160079868b716d5", size = 456398, upload-time = "2026-05-18T04:30:13.784Z" }, + { url = "https://files.pythonhosted.org/packages/0d/64/982ef4a4e5bab5b6e5b6becc8cd5e732f6130a78b855f0abec6439a9a135/watchfiles-1.2.0-cp312-cp312-manylinux_2_31_riscv64.whl", hash = "sha256:d20029a60a71a052a24c4db7673bc4de39ab89adbaccbfb5d67987c5d73f424d", size = 465140, upload-time = "2026-05-18T04:31:52.111Z" }, + { url = "https://files.pythonhosted.org/packages/a0/0c/95282abf4ed680b6096010bcfc30c5fa7a041fc5aa5a2ad17a2cc6c75bba/watchfiles-1.2.0-cp312-cp312-musllinux_1_1_aarch64.whl", hash = "sha256:2cb93af48550faf1cea04c303107c8b75833de7013e57ce27d3b8d21d8d0f58c", size = 630259, upload-time = "2026-05-18T04:31:25.676Z" }, + { url = "https://files.pythonhosted.org/packages/30/45/607c1de1530c4bdcf2cf1d1ecc2505ddba5d96bd43ba9f2b0e79876f850f/watchfiles-1.2.0-cp312-cp312-musllinux_1_1_x86_64.whl", hash = "sha256:2995c176de7692b86a2e4c58d9ec718f753150a979cb4a754e2b4ffa38e70906", size = 659859, upload-time = "2026-05-18T04:30:24.333Z" }, + { url = "https://files.pythonhosted.org/packages/fa/08/d9e2e0f9e8e6791d33aefc694ad7eefa7f901f63caff84a81ded38692f9c/watchfiles-1.2.0-cp312-cp312-win32.whl", hash = "sha256:7a2cffd17d27d2ecbb310c2b1d8174f222a5495b1a721894afa88ec11e25b898", size = 275480, upload-time = "2026-05-18T04:30:31.307Z" }, + { url = "https://files.pythonhosted.org/packages/1c/e6/9d42569c0102645cc8cea5d8c7d8a1e9d4ada2cb7f05f75e554b8aa2202a/watchfiles-1.2.0-cp312-cp312-win_amd64.whl", hash = "sha256:f155b3a1b2a5fc89cdc70d47ee5d54e3b75e88efa34982028a35daef9ba00379", size = 288718, upload-time = "2026-05-18T04:32:10.745Z" }, + { url = "https://files.pythonhosted.org/packages/0a/26/88e0dc6ee3898169d7fa22bb6a69cabf2502d2ee25cb8c876d1262d204f8/watchfiles-1.2.0-cp312-cp312-win_arm64.whl", hash = "sha256:8fa585ede612ee9f9e91b18bebf9ba11b9ae29a4e3a0d0cf6fca3e382133f0d5", size = 281026, upload-time = "2026-05-18T04:30:22.23Z" }, + { url = "https://files.pythonhosted.org/packages/d1/4d/70a7feced9f87e2ff26dba42667290f41694fc64646c67261fbb8cab5d5c/watchfiles-1.2.0-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:01ea8d66f0693b9b60a6541c8d10263091ca9a9060d242f3c1f3143f9aad2c98", size = 399730, upload-time = "2026-05-18T04:31:38.162Z" }, + { url = "https://files.pythonhosted.org/packages/31/3a/0da302f2307aee316922806ebd5726c542cbd787c938271cf14a074c7daf/watchfiles-1.2.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:7ba0480b9a74af058f43b337e937a451e109295c420916d68ad24e3dc02f5e44", size = 392842, upload-time = "2026-05-18T04:30:27.051Z" }, + { url = "https://files.pythonhosted.org/packages/db/ef/d5bdb705c224dbc256aa0c1ec47bf4e61ec52558f2afb44a71a1fe4d7015/watchfiles-1.2.0-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4f34e26a19f91f710c08e0183429f0d1d15df734e6bc78c31e77b9ea9c433658", size = 452989, upload-time = "2026-05-18T04:31:11.945Z" }, + { url = "https://files.pythonhosted.org/packages/71/29/5495f2c1661949ef7a35e4d71111d129cfe7606414a26887a919d0a55406/watchfiles-1.2.0-cp313-cp313-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b4e77f6a55f858504069abd35d336a637555c09bca453dde1ee1e5ada8a6a1fb", size = 458978, upload-time = "2026-05-18T04:30:52.606Z" }, + { url = "https://files.pythonhosted.org/packages/d5/8c/7f9c07c433811c2fffd93e13fdfb7135de9aab5f2ae41be08960fa0047dc/watchfiles-1.2.0-cp313-cp313-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:0cb4d80e212f116474a545c21c912b445f16bb0cef9e6a73a498164223e14e2f", size = 490248, upload-time = "2026-05-18T04:31:36.003Z" }, + { url = "https://files.pythonhosted.org/packages/3c/11/d93632febc52fbc21be90231bb7c17fd5387f46c9076fd40a5f9c2ae6910/watchfiles-1.2.0-cp313-cp313-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:b974946a10af379d425e2eef5b62f5c6ebeaccf91d45eaad6f5b27ecd4f91aa0", size = 571847, upload-time = "2026-05-18T04:31:10.862Z" }, + { url = "https://files.pythonhosted.org/packages/55/b4/383173e73aabb07ad1d9c7aa859d95437ac46a6d6a1e11005facda0c9d19/watchfiles-1.2.0-cp313-cp313-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:86bc13c25a8d1fcd70b51d0ce7c9b65e90de5666fcbfd3e34957cc73ee19aeb5", size = 465974, upload-time = "2026-05-18T04:30:17.006Z" }, + { url = "https://files.pythonhosted.org/packages/a7/6c/89b1a230a78f57c52dd8893adb1f92f94411721b6ec12596c56d98c74356/watchfiles-1.2.0-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ca148d73dea36c9763aaa351e4d7a51780ec1584217c45276f4fe8239c768b71", size = 454782, upload-time = "2026-05-18T04:30:35.656Z" }, + { url = "https://files.pythonhosted.org/packages/24/62/1732118367cfff0a9fce3bf62ff4bfded09ef5df21d9d446b858b3f70a96/watchfiles-1.2.0-cp313-cp313-manylinux_2_31_riscv64.whl", hash = "sha256:c525543d91961c6955b2636b308569e84a1d1c5f5f2932041ab9ef46422f43e3", size = 465182, upload-time = "2026-05-18T04:30:20.846Z" }, + { url = "https://files.pythonhosted.org/packages/28/96/716f7e5f51339bf22963f3345f9f27d7f3b30e2eadc597e257c881dd3c53/watchfiles-1.2.0-cp313-cp313-musllinux_1_1_aarch64.whl", hash = "sha256:a204794696ffb8f9b10fba6f7cb5216d42f3b2b71860ccac6b6e42f5f10973b0", size = 629841, upload-time = "2026-05-18T04:31:05.397Z" }, + { url = "https://files.pythonhosted.org/packages/4c/fe/c40783950fd771ccf66ab3ec2722d188a9af1c7f96c6e811f36e40c6e03f/watchfiles-1.2.0-cp313-cp313-musllinux_1_1_x86_64.whl", hash = "sha256:10d86db20695afe7997ac9e1717637d6714a8d0220458c33f3d2061f54cec427", size = 658028, upload-time = "2026-05-18T04:31:48.22Z" }, + { url = "https://files.pythonhosted.org/packages/71/72/4508db1856d1d87fcbb3b63f4839bab1b5682cb0e8d224d122263c09654a/watchfiles-1.2.0-cp313-cp313-win32.whl", hash = "sha256:eb283ee99e21ad6443c8cdb06ac5b34b1308c329cbdf03fa02b445363714c799", size = 275183, upload-time = "2026-05-18T04:30:59.57Z" }, + { url = "https://files.pythonhosted.org/packages/f9/36/14b76ca57652e5cc5fd1c11f32a261292c08a0d19a00351013c2549cbfb2/watchfiles-1.2.0-cp313-cp313-win_amd64.whl", hash = "sha256:a0f27f01bee51861392bb6b7c4fdb290b27d1eb194e9e28788d68102a0e898d9", size = 288059, upload-time = "2026-05-18T04:32:07.937Z" }, + { url = "https://files.pythonhosted.org/packages/1b/8d/0a85e395398d8d20fadfe5c5d32c726eee17a519e78fb356f2cf7531bffe/watchfiles-1.2.0-cp313-cp313-win_arm64.whl", hash = "sha256:3651aa7058595e9cfb75d35dd5ada2bf9f48a5b8a0f3562821d3e210c507e077", size = 280186, upload-time = "2026-05-18T04:31:54.484Z" }, + { url = "https://files.pythonhosted.org/packages/37/68/36db056f1fdcc5f07302f56e631774d6835bcd6fa3ace402304621d5f9e5/watchfiles-1.2.0-cp313-cp313t-macosx_10_12_x86_64.whl", hash = "sha256:faea288b6f0ab1902ef08f4ca6de005dccf856c4e0c4f21b8c5fce02d90a1b08", size = 399031, upload-time = "2026-05-18T04:30:44.576Z" }, + { url = "https://files.pythonhosted.org/packages/c1/64/01a9d6f66a82a5c101ce939274106cc72759d62427e153f01edd2b9f87c2/watchfiles-1.2.0-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:01859b11fd9fbca670f4d5da00fbac282cfea9bd67a2125d8b2833a3b5617ea9", size = 391205, upload-time = "2026-05-18T04:30:25.413Z" }, + { url = "https://files.pythonhosted.org/packages/84/2c/0a44fe058cb4bb7b8ede6b6670698bbb7c0400740e378d00022189b7b31d/watchfiles-1.2.0-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:fff610d7bb2256a317bb1e96f0d7862c7aa8076733ee5df0fd41bbe76a24a4f4", size = 451892, upload-time = "2026-05-18T04:32:14.005Z" }, + { url = "https://files.pythonhosted.org/packages/67/a1/351e0d56cd35e6488b5c8b4fb11a809a5bc923e8fe8fed9faf8920be0c89/watchfiles-1.2.0-cp313-cp313t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b141a4891c995a039cd89e9a49e62df1dc8a559a5d1a6e4c7106d16c12777a55", size = 458867, upload-time = "2026-05-18T04:31:22.279Z" }, + { url = "https://files.pythonhosted.org/packages/d5/7d/9d09605187f1b838998624049fcf8bf47b73c1a3b76901fcac1782f62277/watchfiles-1.2.0-cp313-cp313t-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:f22943b7770483f6ea0721c6b11d022947a98eb0acae14694de034f4d0d38925", size = 490217, upload-time = "2026-05-18T04:31:43.657Z" }, + { url = "https://files.pythonhosted.org/packages/60/5d/a17a16eccb182f04188cd308ec24b1a71a9b5c4e7098269cf35d9fa56d02/watchfiles-1.2.0-cp313-cp313t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:1bc6195825b7dcd217968bb1f801a60fd4c16e8eeab5bedc7fe917d7d5995ab4", size = 571458, upload-time = "2026-05-18T04:32:11.875Z" }, + { url = "https://files.pythonhosted.org/packages/d3/3d/4dd457062083ab1938e5dfd45032eb425cee2ac817287ca8ff4356183e5d/watchfiles-1.2.0-cp313-cp313t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:d4a4b147f5dca2a5d325a06a832fb43f345751adfbc63204aec30e0d9ca965a2", size = 464707, upload-time = "2026-05-18T04:30:43.492Z" }, + { url = "https://files.pythonhosted.org/packages/c6/71/ea8c57b128f5383de74d0c7d2d9c57ad7c9a65a930c451bd25d524b295b7/watchfiles-1.2.0-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:4543579a9bdb0c9560039b4ffddbdb39545707659fbc430ce4c10f3f68d557f9", size = 454663, upload-time = "2026-05-18T04:30:16.061Z" }, + { url = "https://files.pythonhosted.org/packages/53/fd/2e812bf938406d7db351f0703ddd3fc6c061cf30d96153a77bc79a943a44/watchfiles-1.2.0-cp313-cp313t-manylinux_2_31_riscv64.whl", hash = "sha256:20aa0e708b920bde876a4aa82dc7dd6ebea228a63a67cda6632c2fc87b787efa", size = 463537, upload-time = "2026-05-18T04:31:44.9Z" }, + { url = "https://files.pythonhosted.org/packages/86/56/d17a7f1dd1bc3035f1072694a551301272f1739c2d8e319c927cb9e29b38/watchfiles-1.2.0-cp313-cp313t-musllinux_1_1_aarch64.whl", hash = "sha256:d413349d565dab74297f2a63e84a097936be69bf8f3b3801f27f380e32040f44", size = 629194, upload-time = "2026-05-18T04:31:14.141Z" }, + { url = "https://files.pythonhosted.org/packages/be/06/f1ff66bf5cae50aa4062779a0ecd0bbaf15e466195719074078947d9a17d/watchfiles-1.2.0-cp313-cp313t-musllinux_1_1_x86_64.whl", hash = "sha256:f28b2725eb8cce327b9b3ab02415c853011dc55c95832fe90de6bc56f5315f72", size = 656194, upload-time = "2026-05-18T04:31:47.14Z" }, + { url = "https://files.pythonhosted.org/packages/23/f4/7513ef1e85fc4c6331b59479d6d72661fc391fbe543678052ac72c8b6c19/watchfiles-1.2.0-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:4674d49eb94706dfe666c069fc0a1b646ffcf920473492e209f6d5f60d3f0cc2", size = 403050, upload-time = "2026-05-18T04:30:36.753Z" }, + { url = "https://files.pythonhosted.org/packages/27/0b/a54103cfd732bb703c7a749222011a0483ef3705948dae3b203158601119/watchfiles-1.2.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:094b9b70103d4e963499bdea001ee3c2697b144cd9ae6218a62c0f89ec9e31db", size = 396629, upload-time = "2026-05-18T04:32:03.268Z" }, + { url = "https://files.pythonhosted.org/packages/5e/2c/73f31a3b893886206c3f54d73e8ad8dee58cdb2f69ad2622e0a8a9e07f4e/watchfiles-1.2.0-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:b0ef001f8c25ad0fa9529f914c1600647ecd0f542d11c19b7894768c67b6acb7", size = 457318, upload-time = "2026-05-18T04:31:01.932Z" }, + { url = "https://files.pythonhosted.org/packages/e9/f9/45d021e4a5cc7b9dd567f7cbb06d3b75f751a690063fb6cc7ec60f4e46b7/watchfiles-1.2.0-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a88fc94e647bc4eec523f1caa540258eb71d14278b9daf72fa1e2658a98df0f0", size = 457771, upload-time = "2026-05-18T04:30:56.331Z" }, +] + [[package]] name = "websockets" version = "15.0.1"