agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,4167 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import array
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import math
|
|
7
|
+
import logging
|
|
8
|
+
import os
|
|
9
|
+
import re
|
|
10
|
+
import threading
|
|
11
|
+
import time
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from urllib.parse import urlsplit
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from collections.abc import Awaitable, Callable
|
|
16
|
+
from typing import Any, AsyncIterable
|
|
17
|
+
from uuid import uuid4
|
|
18
|
+
|
|
19
|
+
try:
|
|
20
|
+
from livekit import api, rtc
|
|
21
|
+
from livekit.agents import (
|
|
22
|
+
Agent,
|
|
23
|
+
AgentSession,
|
|
24
|
+
AudioConfig,
|
|
25
|
+
BackgroundAudioPlayer,
|
|
26
|
+
RunContext,
|
|
27
|
+
function_tool,
|
|
28
|
+
metrics,
|
|
29
|
+
)
|
|
30
|
+
from livekit.agents.utils.audio import audio_frames_from_file
|
|
31
|
+
from livekit.agents.voice.background_audio import BuiltinAudioClip
|
|
32
|
+
from livekit.agents.types import (
|
|
33
|
+
ATTRIBUTE_TRANSCRIPTION_TRACK_ID,
|
|
34
|
+
TOPIC_TRANSCRIPTION,
|
|
35
|
+
)
|
|
36
|
+
from livekit.agents.voice import ModelSettings
|
|
37
|
+
from livekit.agents.voice.io import TimedString
|
|
38
|
+
from livekit.agents.voice.room_io import AudioInputOptions, RoomOptions
|
|
39
|
+
from livekit.api import AccessToken, VideoGrants
|
|
40
|
+
from livekit.plugins import silero
|
|
41
|
+
from livekit.protocol.sip import (
|
|
42
|
+
CreateSIPDispatchRuleRequest,
|
|
43
|
+
DeleteSIPDispatchRuleRequest,
|
|
44
|
+
ListSIPDispatchRuleRequest,
|
|
45
|
+
SIPDispatchRule,
|
|
46
|
+
SIPDispatchRuleDirect,
|
|
47
|
+
)
|
|
48
|
+
except ImportError as exc:
|
|
49
|
+
raise ImportError(
|
|
50
|
+
"LiveKit mode requires the 'livekit' optional dependency"
|
|
51
|
+
) from exc
|
|
52
|
+
|
|
53
|
+
from datetime import datetime, timezone
|
|
54
|
+
|
|
55
|
+
from fi.simulate._logging import redacted_exc_info
|
|
56
|
+
from fi.simulate.agent.definition import (
|
|
57
|
+
AgentDefinition,
|
|
58
|
+
LiveKitSimulatorRuntime,
|
|
59
|
+
LLMConfig,
|
|
60
|
+
ProviderEvidenceConfig,
|
|
61
|
+
RetellTargetConfig,
|
|
62
|
+
SimulatorAgentDefinition,
|
|
63
|
+
STTConfig,
|
|
64
|
+
TelephonyTransport,
|
|
65
|
+
TTSConfig,
|
|
66
|
+
VapiTargetConfig,
|
|
67
|
+
VoiceProviderTarget,
|
|
68
|
+
)
|
|
69
|
+
from fi.simulate.artifacts.manifest import ArtifactManifestEntry
|
|
70
|
+
from fi.simulate.evidence.base import EvidenceSourceSummary
|
|
71
|
+
from fi.simulate.evidence.providers import (
|
|
72
|
+
EvidenceContext,
|
|
73
|
+
ProviderConfigError,
|
|
74
|
+
ProviderFetchResult,
|
|
75
|
+
RetellEvidenceSource,
|
|
76
|
+
VapiEvidenceSource,
|
|
77
|
+
)
|
|
78
|
+
from fi.simulate.endpoints.originators import (
|
|
79
|
+
CallOriginator,
|
|
80
|
+
build_call_originator,
|
|
81
|
+
finalize_originator,
|
|
82
|
+
)
|
|
83
|
+
from fi.simulate.simulation.bridge import LiveKitAudioBridge
|
|
84
|
+
from fi.simulate.simulation.bridge.audio import PCMResampler
|
|
85
|
+
from fi.simulate.simulation.livekit_models import LiveKitModels, build_livekit_models
|
|
86
|
+
from fi.simulate.recording.room_recorder import (
|
|
87
|
+
RoomRecorder,
|
|
88
|
+
mix_recordings,
|
|
89
|
+
mix_recordings_stereo,
|
|
90
|
+
)
|
|
91
|
+
from fi.simulate.runtime import (
|
|
92
|
+
FailureStage,
|
|
93
|
+
SimulationFailure,
|
|
94
|
+
TestCaseStatus,
|
|
95
|
+
derive_test_case_id,
|
|
96
|
+
new_run_id,
|
|
97
|
+
)
|
|
98
|
+
from fi.simulate.simulation.engines.base import BaseEngine
|
|
99
|
+
from fi.simulate.simulation.generator import ScenarioGenerator
|
|
100
|
+
from fi.simulate.simulation.models import Persona, Scenario, TestCaseResult, TestReport
|
|
101
|
+
from fi.simulate.simulation.voice_prompt import CallType, build_voice_simulator_prompt
|
|
102
|
+
|
|
103
|
+
logger = logging.getLogger(__name__)
|
|
104
|
+
_SAFE_ROOM = re.compile(r"[^A-Za-z0-9_.-]+")
|
|
105
|
+
# On conversation end, wait up to this long for the party still finishing its
|
|
106
|
+
# own turn to commit it (a LiveKit turn lands in history only after its TTS
|
|
107
|
+
# finishes playing), then delete the room so neither side keeps talking into a
|
|
108
|
+
# call the other has already left.
|
|
109
|
+
_FINAL_TURN_COMMIT_WAIT_SECONDS = 30.0
|
|
110
|
+
# The hosted platform inflates ``cleanup_timeout`` to carry the whole run
|
|
111
|
+
# budget (observed 1470s); as a per-step cleanup bound it must stay capped.
|
|
112
|
+
_MAX_CLEANUP_TIMEOUT_SECONDS = 60.0
|
|
113
|
+
# A dead LiveKit signal connection can leave any one SDK cleanup await pending
|
|
114
|
+
# indefinitely. The case-level deadline is still the outer bound, but no
|
|
115
|
+
# single best-effort operation may consume it all and starve every cleanup that
|
|
116
|
+
# follows. Session close gets longer because it drains several SDK activities.
|
|
117
|
+
_CLEANUP_STEP_TIMEOUT_SECONDS = 8.0
|
|
118
|
+
_SESSION_CLEANUP_TIMEOUT_SECONDS = 15.0
|
|
119
|
+
_BACKGROUND_AUDIO_CLEANUP_TIMEOUT_SECONDS = 5.0
|
|
120
|
+
_NO_CONVERSATION_TIMEOUT_SECONDS = 120.0
|
|
121
|
+
# How long the side that was meant to speak first is given before the simulated person speaks
|
|
122
|
+
# instead. Both sides are voice agents waiting to be addressed, so when the one that placed the call
|
|
123
|
+
# says nothing the call is silence until a deadline discards it, and nothing was learned about
|
|
124
|
+
# either side. Kept well under the timeout above, which is what abandons a call nobody started.
|
|
125
|
+
#
|
|
126
|
+
# Eight seconds: the smallest bound that cannot pre-empt a slow first turn.
|
|
127
|
+
_OPEN_INSTEAD_AFTER_SECONDS = 8.0
|
|
128
|
+
# Frequency and length per kind of mailbox. FULL has no entry: it invites no message.
|
|
129
|
+
_VOICEMAIL_TONE_BY_STYLE: dict[str, tuple[float, float]] = {
|
|
130
|
+
"personal": (1000.0, 0.40),
|
|
131
|
+
"carrier": (1400.0, 0.33),
|
|
132
|
+
"operator": (440.0, 0.52),
|
|
133
|
+
}
|
|
134
|
+
_DEFAULT_VOICEMAIL_STYLE = "personal"
|
|
135
|
+
# Loud enough to be unmistakable against speech, which reaches 15000 to 23000 of 32768.
|
|
136
|
+
_VOICEMAIL_TONE_VOLUME = 0.8
|
|
137
|
+
# A mailbox plays one greeting and then records, so the eight-message conversation floor is
|
|
138
|
+
# unreachable however well the agent behaves, and holding it there errored every voicemail call.
|
|
139
|
+
_VOICEMAIL_MIN_TURN_MESSAGES = 1
|
|
140
|
+
# How long a mailbox records before cutting the line, from the end of the tone or greeting. Without a
|
|
141
|
+
# bound the call runs to the silence watchdog with the agent still talking into a machine.
|
|
142
|
+
_VOICEMAIL_RECORD_SECONDS = 40.0
|
|
143
|
+
# Resamplers into the mixer's rate, one per source rate, kept because ``ratecv`` is stateful.
|
|
144
|
+
_MIXER_RESAMPLERS: dict[tuple[int, int], PCMResampler] = {}
|
|
145
|
+
# How long after the mailbox stops speaking the tone comes. A real system leaves a beat.
|
|
146
|
+
_VOICEMAIL_TONE_GAP_SECONDS = 0.7
|
|
147
|
+
# How long to wait for the mailbox to say anything before giving up on the tone. Bounded so a
|
|
148
|
+
# mailbox that never speaks cannot leave this task pending for the length of the call.
|
|
149
|
+
_VOICEMAIL_TONE_WAIT_SECONDS = 40.0
|
|
150
|
+
# The mixer reinterprets frames at this rate rather than resampling them, so anything published
|
|
151
|
+
# through it must be produced here or it plays at the wrong pitch and length.
|
|
152
|
+
_BACKGROUND_MIXER_RATE = 48000
|
|
153
|
+
# Each web case drives a full voice pipeline (STT/LLM/TTS + LiveKit conns) in one
|
|
154
|
+
# child; too many starve the pod's CPU. This is an OPS CEILING on the
|
|
155
|
+
# config-driven ``max_parallel_cases`` (not a replacement for it) — tune
|
|
156
|
+
# ``ALK_VOICE_MAX_CASE_CONCURRENCY`` to the pod's cores. Caps web cases only.
|
|
157
|
+
_VOICE_MAX_CASE_CONCURRENCY_DEFAULT = 4
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _simulator_participant_identity(persona: Persona, test_case_id: str) -> str:
|
|
161
|
+
"""Give repository agents the scenario caller ANI through a standard identity seam.
|
|
162
|
+
|
|
163
|
+
LiveKit token metadata is not exposed consistently across every SDK/agent version. The
|
|
164
|
+
harness therefore uses the identity convention already understood by repository voice
|
|
165
|
+
agents: ``fagi-simulator-phone-<digits>-...``. A persona without a fixture-derived phone
|
|
166
|
+
keeps the legacy anonymous identity.
|
|
167
|
+
"""
|
|
168
|
+
definition = persona.persona if isinstance(persona.persona, dict) else {}
|
|
169
|
+
metadata = definition.get("metadata")
|
|
170
|
+
metadata = metadata if isinstance(metadata, dict) else {}
|
|
171
|
+
digits = re.sub(r"\D", "", str(metadata.get("caller_phone") or ""))
|
|
172
|
+
suffix = test_case_id[-12:]
|
|
173
|
+
if 7 <= len(digits) <= 15:
|
|
174
|
+
return f"fagi-simulator-phone-{digits}-{suffix}"
|
|
175
|
+
return f"fagi-simulator-{suffix}"
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _voice_max_case_concurrency() -> int:
|
|
179
|
+
raw = os.environ.get("ALK_VOICE_MAX_CASE_CONCURRENCY", "").strip()
|
|
180
|
+
if not raw:
|
|
181
|
+
return _VOICE_MAX_CASE_CONCURRENCY_DEFAULT
|
|
182
|
+
try:
|
|
183
|
+
value = int(raw)
|
|
184
|
+
except ValueError:
|
|
185
|
+
return _VOICE_MAX_CASE_CONCURRENCY_DEFAULT
|
|
186
|
+
return value if value >= 1 else _VOICE_MAX_CASE_CONCURRENCY_DEFAULT
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
_silero_vad: Any | None = None
|
|
190
|
+
_silero_vad_guard = threading.Lock()
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _load_silero_vad_sync() -> Any:
|
|
194
|
+
"""One shared VAD per process; per-case ``VAD.load()`` ran a synchronous
|
|
195
|
+
model load on the event loop for every concurrent case."""
|
|
196
|
+
global _silero_vad
|
|
197
|
+
with _silero_vad_guard:
|
|
198
|
+
if _silero_vad is None:
|
|
199
|
+
_silero_vad = silero.VAD.load()
|
|
200
|
+
return _silero_vad
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
@dataclass(frozen=True)
|
|
204
|
+
class _TargetParticipant:
|
|
205
|
+
identity: str
|
|
206
|
+
sid: str
|
|
207
|
+
audio_track_sid: str
|
|
208
|
+
attributes: dict[str, str] = field(default_factory=dict)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
@dataclass
|
|
212
|
+
class _CaseOutcome:
|
|
213
|
+
status: TestCaseStatus
|
|
214
|
+
transcript: str = ""
|
|
215
|
+
messages: list[dict[str, str]] = field(default_factory=list)
|
|
216
|
+
failure: SimulationFailure | None = None
|
|
217
|
+
audio_input_path: str | None = None
|
|
218
|
+
audio_output_path: str | None = None
|
|
219
|
+
audio_combined_path: str | None = None
|
|
220
|
+
audio_stereo_path: str | None = None
|
|
221
|
+
metadata: dict[str, object] = field(default_factory=dict)
|
|
222
|
+
evidence: list[EvidenceSourceSummary] = field(default_factory=list)
|
|
223
|
+
provider_artifacts: list[ArtifactManifestEntry] = field(default_factory=list)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _dispatch_metadata_json(agent_definition) -> str:
|
|
227
|
+
"""Metadata for the target agent's LiveKit dispatch.
|
|
228
|
+
|
|
229
|
+
EMPTY by default: a target agent built from a LiveKit template branches on
|
|
230
|
+
``ctx.job.metadata`` and treats any non-empty payload as an outbound/no-greet
|
|
231
|
+
job, so it never publishes an audio track and readiness times out
|
|
232
|
+
(``agent_unavailable``). Only a target explicitly built to consume dispatch
|
|
233
|
+
metadata sets ``agent_definition.dispatch_metadata``.
|
|
234
|
+
"""
|
|
235
|
+
meta = getattr(agent_definition, "dispatch_metadata", None)
|
|
236
|
+
return json.dumps(meta, sort_keys=True) if meta else ""
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _resolve_target_profile(kind: str):
|
|
240
|
+
"""Look up the target adapter's profile — the factory that replaced the
|
|
241
|
+
engine's ``transport.kind`` branching. Unknown kinds fail loudly, which is
|
|
242
|
+
what makes it safe to open ``TelephonyTransport.kind`` from a Literal to a
|
|
243
|
+
free string later."""
|
|
244
|
+
from fi.simulate.endpoints.profiles import get_profile
|
|
245
|
+
|
|
246
|
+
profile = get_profile(kind)
|
|
247
|
+
if profile is None:
|
|
248
|
+
raise ValueError(f"unsupported_transport_kind: {kind}")
|
|
249
|
+
return profile
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _simulator_turn_handling(
|
|
253
|
+
*,
|
|
254
|
+
vad: object | None,
|
|
255
|
+
allow_interruptions: bool | None = None,
|
|
256
|
+
min_endpointing_delay: float | None = None,
|
|
257
|
+
max_endpointing_delay: float | None = None,
|
|
258
|
+
) -> dict[str, object]:
|
|
259
|
+
return {
|
|
260
|
+
"turn_detection": "vad" if vad is not None else "stt",
|
|
261
|
+
# A short delay fires inside a sentence, on a comma or a breath, so the caller treats a pause
|
|
262
|
+
# as the end of the turn, talks over the agent and then repeats itself for want of an answer.
|
|
263
|
+
"endpointing": {
|
|
264
|
+
"mode": "fixed",
|
|
265
|
+
"min_delay": min_endpointing_delay or 0.9,
|
|
266
|
+
"max_delay": max_endpointing_delay or 3.0,
|
|
267
|
+
},
|
|
268
|
+
# A real caller interrupts, but only over something long enough to be worth interrupting.
|
|
269
|
+
"interruption": {
|
|
270
|
+
"enabled": (True if allow_interruptions is None else allow_interruptions),
|
|
271
|
+
"discard_audio_if_uninterruptible": True,
|
|
272
|
+
"min_duration": 0.6,
|
|
273
|
+
},
|
|
274
|
+
"preemptive_generation": {"enabled": True},
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
class _TestRunnerAgent(Agent):
|
|
279
|
+
def __init__(
|
|
280
|
+
self,
|
|
281
|
+
persona: Persona,
|
|
282
|
+
*,
|
|
283
|
+
min_turn_messages: int = 0,
|
|
284
|
+
**kwargs,
|
|
285
|
+
):
|
|
286
|
+
turn_handling = kwargs.setdefault(
|
|
287
|
+
"turn_handling",
|
|
288
|
+
_simulator_turn_handling(vad=kwargs.get("vad")),
|
|
289
|
+
)
|
|
290
|
+
super().__init__(**kwargs)
|
|
291
|
+
self._persona = persona
|
|
292
|
+
self._min_turn_messages = min_turn_messages
|
|
293
|
+
self._session_turn_handling = turn_handling
|
|
294
|
+
self._session: AgentSession | None = None
|
|
295
|
+
self._end_requested = asyncio.Event()
|
|
296
|
+
self._end_speech_handle: Any | None = None
|
|
297
|
+
self._usage_collector = metrics.ModelUsageCollector()
|
|
298
|
+
|
|
299
|
+
@function_tool(
|
|
300
|
+
name="endCall",
|
|
301
|
+
# Nothing quotable and nothing English-specific: wording here comes back out as speech.
|
|
302
|
+
description=(
|
|
303
|
+
"Ends the call. Nothing else ends it and no one else ends it for you. "
|
|
304
|
+
"Use it once you have nothing further."
|
|
305
|
+
),
|
|
306
|
+
)
|
|
307
|
+
async def end_call(self, ctx: RunContext) -> str:
|
|
308
|
+
if self._session is None:
|
|
309
|
+
logger.warning("endCall refused: no session yet")
|
|
310
|
+
return "Continue the conversation before ending the call."
|
|
311
|
+
messages = _session_messages(self._session)
|
|
312
|
+
floor, alternation_required = _turn_requirements(self._min_turn_messages)
|
|
313
|
+
below_floor = len(messages) < floor or (
|
|
314
|
+
alternation_required and not _has_role_alternation(messages)
|
|
315
|
+
)
|
|
316
|
+
if below_floor and _target_has_gone_quiet(messages):
|
|
317
|
+
below_floor = False
|
|
318
|
+
if below_floor:
|
|
319
|
+
# Whether the caller ever reached for this tool, and why it was turned away, is the
|
|
320
|
+
# difference between a simulator that will not hang up and one that was not allowed to.
|
|
321
|
+
logger.warning(
|
|
322
|
+
"endCall refused: %d messages, floor %d, alternating=%s",
|
|
323
|
+
len(messages),
|
|
324
|
+
floor,
|
|
325
|
+
_has_role_alternation(messages),
|
|
326
|
+
)
|
|
327
|
+
# "Not yet" rather than "stop asking", or the caller never retries the tool.
|
|
328
|
+
return (
|
|
329
|
+
f"Not yet: {len(messages)} of {floor} messages so far and both speakers must "
|
|
330
|
+
"have spoken. Keep the conversation going, then call endCall again."
|
|
331
|
+
)
|
|
332
|
+
logger.warning("endCall accepted after %d messages", len(messages))
|
|
333
|
+
# The tool runs inside the same SpeechHandle that carries the model's
|
|
334
|
+
# natural closing sentence. Remember that exact handle before waking
|
|
335
|
+
# the outer runner so it cannot snapshot history in the brief interval
|
|
336
|
+
# before TTS starts and ``session.current_speech`` becomes non-None.
|
|
337
|
+
self._end_speech_handle = ctx.speech_handle
|
|
338
|
+
self._end_requested.set()
|
|
339
|
+
return "Conversation ended."
|
|
340
|
+
|
|
341
|
+
async def wait_for_end_speech(self) -> None:
|
|
342
|
+
if self._end_speech_handle is not None:
|
|
343
|
+
await self._end_speech_handle
|
|
344
|
+
|
|
345
|
+
@property
|
|
346
|
+
def started_session(self) -> AgentSession | None:
|
|
347
|
+
return self._session
|
|
348
|
+
|
|
349
|
+
@property
|
|
350
|
+
def end_requested(self) -> asyncio.Event:
|
|
351
|
+
return self._end_requested
|
|
352
|
+
|
|
353
|
+
@property
|
|
354
|
+
def model_usage(self) -> list[dict[str, object]]:
|
|
355
|
+
return [
|
|
356
|
+
usage.model_dump(mode="json")
|
|
357
|
+
for usage in sorted(
|
|
358
|
+
self._usage_collector.flatten(),
|
|
359
|
+
key=lambda usage: (usage.type, usage.provider, usage.model),
|
|
360
|
+
)
|
|
361
|
+
]
|
|
362
|
+
|
|
363
|
+
async def start_session(
|
|
364
|
+
self,
|
|
365
|
+
room: rtc.Room,
|
|
366
|
+
*,
|
|
367
|
+
participant_kinds: list | None = None,
|
|
368
|
+
participant_identity: str | None = None,
|
|
369
|
+
) -> AgentSession:
|
|
370
|
+
session = AgentSession(
|
|
371
|
+
stt=self.stt,
|
|
372
|
+
llm=self.llm,
|
|
373
|
+
tts=self.tts,
|
|
374
|
+
vad=self.vad,
|
|
375
|
+
turn_handling=self._session_turn_handling,
|
|
376
|
+
)
|
|
377
|
+
self._session = session
|
|
378
|
+
session.on(
|
|
379
|
+
"metrics_collected",
|
|
380
|
+
lambda event: self._usage_collector.collect(event.metrics),
|
|
381
|
+
)
|
|
382
|
+
default_kinds = [
|
|
383
|
+
rtc.ParticipantKind.PARTICIPANT_KIND_STANDARD,
|
|
384
|
+
getattr(
|
|
385
|
+
rtc.ParticipantKind,
|
|
386
|
+
"PARTICIPANT_KIND_AGENT",
|
|
387
|
+
rtc.ParticipantKind.PARTICIPANT_KIND_STANDARD,
|
|
388
|
+
),
|
|
389
|
+
rtc.ParticipantKind.PARTICIPANT_KIND_SIP,
|
|
390
|
+
]
|
|
391
|
+
room_kwargs: dict = {
|
|
392
|
+
"audio_input": AudioInputOptions(
|
|
393
|
+
pre_connect_audio=False,
|
|
394
|
+
pre_connect_audio_timeout=3.0,
|
|
395
|
+
),
|
|
396
|
+
# Enabled to build RoomIO's TranscriptSynchronizer, which aligns the
|
|
397
|
+
# spoken transcript to audio playback. On an interruption the
|
|
398
|
+
# recorded turn is then truncated to what was actually said instead
|
|
399
|
+
# of the full LLM text (playback-timing estimate — works with any
|
|
400
|
+
# TTS, unlike use_tts_aligned_transcript which needs word timing our
|
|
401
|
+
# Deepgram/Gemini voices don't emit and would drop the turn). The
|
|
402
|
+
# simulator's transcription is published to the room as a harmless
|
|
403
|
+
# side effect (our target-transcription handler filters by identity).
|
|
404
|
+
"text_output": True,
|
|
405
|
+
"close_on_disconnect": False,
|
|
406
|
+
"delete_room_on_close": False,
|
|
407
|
+
"participant_kinds": participant_kinds or default_kinds,
|
|
408
|
+
}
|
|
409
|
+
if participant_identity:
|
|
410
|
+
room_kwargs["participant_identity"] = participant_identity
|
|
411
|
+
await session.start(
|
|
412
|
+
self,
|
|
413
|
+
room=room,
|
|
414
|
+
room_options=RoomOptions(**room_kwargs),
|
|
415
|
+
)
|
|
416
|
+
await self._maybe_start_background_audio(room, session)
|
|
417
|
+
return session
|
|
418
|
+
|
|
419
|
+
async def _maybe_start_background_audio(
|
|
420
|
+
self, room: "rtc.Room", session: "AgentSession"
|
|
421
|
+
) -> None:
|
|
422
|
+
"""Mix caller-side ambient noise under the simulated caller, if the run asked for it.
|
|
423
|
+
|
|
424
|
+
Off unless HARNESS_BACKGROUND_NOISE names a source: a LiveKit builtin clip name, or an
|
|
425
|
+
http(s) URL to an ambient file. Any failure is swallowed, because a call without ambience is
|
|
426
|
+
preferable to a dropped one.
|
|
427
|
+
"""
|
|
428
|
+
source = os.environ.get("HARNESS_BACKGROUND_NOISE", "").strip()
|
|
429
|
+
# A mailbox needs this method for its tone and its recording timer, and FULL has no tone.
|
|
430
|
+
tone_style = _voicemail_tone_style()
|
|
431
|
+
if _answered_by_voicemail() and source:
|
|
432
|
+
# Nothing stands behind a recording, and a room behind one gives the game away.
|
|
433
|
+
logger.info("mailbox answered, so ambience is dropped (noise %r)", source)
|
|
434
|
+
source = ""
|
|
435
|
+
if not source and not tone_style and not _answered_by_voicemail():
|
|
436
|
+
return
|
|
437
|
+
|
|
438
|
+
try:
|
|
439
|
+
# 2.0, not the 0.3 this used to default to. Measured in an isolated two-participant
|
|
440
|
+
# room, the office clip peaks at 119 of 32768 at 0.3, which is below the noise floor of
|
|
441
|
+
# speech near 15000: the ambience played and nobody could hear it. At 2.0 the same clip
|
|
442
|
+
# measures 752 to 789 on real calls, which is audible under a voice without masking it.
|
|
443
|
+
volume = float(os.environ.get("HARNESS_BACKGROUND_NOISE_VOLUME", "2.0"))
|
|
444
|
+
clip_source: Any = None
|
|
445
|
+
if source.startswith(("http://", "https://")):
|
|
446
|
+
clip_source = await asyncio.to_thread(_downloaded_audio, source)
|
|
447
|
+
if not clip_source:
|
|
448
|
+
return
|
|
449
|
+
self._background_noise_file = clip_source
|
|
450
|
+
elif source:
|
|
451
|
+
clip_source = getattr(BuiltinAudioClip, source, None)
|
|
452
|
+
if clip_source is None:
|
|
453
|
+
logger.warning(
|
|
454
|
+
"background audio clip %r is not one LiveKit ships", source
|
|
455
|
+
)
|
|
456
|
+
return
|
|
457
|
+
# A player is created even with no ambience clip, because a mailbox tone needs a
|
|
458
|
+
# published track whether or not this scenario also asked for a room.
|
|
459
|
+
player = (
|
|
460
|
+
BackgroundAudioPlayer(
|
|
461
|
+
ambient_sound=AudioConfig(clip_source, volume=volume)
|
|
462
|
+
)
|
|
463
|
+
if clip_source is not None
|
|
464
|
+
else BackgroundAudioPlayer()
|
|
465
|
+
)
|
|
466
|
+
await player.start(room=room, agent_session=session)
|
|
467
|
+
self._background_player = player
|
|
468
|
+
except Exception:
|
|
469
|
+
logger.warning("background audio not started", exc_info=True)
|
|
470
|
+
return
|
|
471
|
+
# Spoken as its own turn, so the transcript shows what the agent heard.
|
|
472
|
+
recorded = os.environ.get("HARNESS_VOICEMAIL_CLIP", "").strip()
|
|
473
|
+
if recorded.startswith(("http://", "https://")):
|
|
474
|
+
# Catalogue clips are meant to be served from object storage rather than shipped in the
|
|
475
|
+
# image, and a URL cannot be decoded in place.
|
|
476
|
+
recorded = await asyncio.to_thread(_downloaded_audio, recorded) or ""
|
|
477
|
+
said = os.environ.get("HARNESS_VOICEMAIL_CLIP_TRANSCRIPT", "").strip()
|
|
478
|
+
if recorded and said:
|
|
479
|
+
try:
|
|
480
|
+
self._voicemail_greeting = session.say(
|
|
481
|
+
said,
|
|
482
|
+
audio=audio_frames_from_file(recorded),
|
|
483
|
+
allow_interruptions=False,
|
|
484
|
+
)
|
|
485
|
+
except Exception:
|
|
486
|
+
logger.warning("recorded mailbox greeting not played", exc_info=True)
|
|
487
|
+
elif recorded:
|
|
488
|
+
# No words for it, so it cannot be a turn: an invented line would put words in the
|
|
489
|
+
# transcript that the audio never says, and an eval would judge those words.
|
|
490
|
+
logger.warning("mailbox clip has no transcript; playing it without a turn")
|
|
491
|
+
try:
|
|
492
|
+
player.play(AudioConfig(recorded, volume=1.0))
|
|
493
|
+
except Exception:
|
|
494
|
+
logger.warning("recorded mailbox greeting not played", exc_info=True)
|
|
495
|
+
if tone_style:
|
|
496
|
+
self._voicemail_tone_task = asyncio.create_task(
|
|
497
|
+
self._play_voicemail_tone(tone_style, session)
|
|
498
|
+
)
|
|
499
|
+
if _answered_by_voicemail():
|
|
500
|
+
self._mailbox_close_task = asyncio.create_task(
|
|
501
|
+
self._close_mailbox_after_recording()
|
|
502
|
+
)
|
|
503
|
+
|
|
504
|
+
_voicemail_greeting: Any = None
|
|
505
|
+
_voicemail_tone_task: Any = None
|
|
506
|
+
_mailbox_close_task: Any = None
|
|
507
|
+
|
|
508
|
+
async def _close_mailbox_after_recording(self) -> None:
|
|
509
|
+
"""Stop recording and cut the line, the way a mailbox does.
|
|
510
|
+
|
|
511
|
+
A mailbox is not a party to the call. It never says goodbye, it never asks whether anybody is
|
|
512
|
+
there, and it does not wait: it records for as long as it records and then hangs up. Nothing
|
|
513
|
+
here was ending these calls, so they ran to the silence watchdog with the agent talking into
|
|
514
|
+
a machine long after it had left its message.
|
|
515
|
+
|
|
516
|
+
The clock starts once the greeting and the tone are done where there are either, and at call
|
|
517
|
+
start otherwise, which is the FULL mailbox: it invites no message, so the window it gets is
|
|
518
|
+
generous rather than precise.
|
|
519
|
+
"""
|
|
520
|
+
try:
|
|
521
|
+
if self._voicemail_greeting is not None:
|
|
522
|
+
await self._voicemail_greeting
|
|
523
|
+
tone = self._voicemail_tone_task
|
|
524
|
+
if tone is not None:
|
|
525
|
+
try:
|
|
526
|
+
await tone
|
|
527
|
+
except Exception:
|
|
528
|
+
# The tone failing is not a reason to record for ever.
|
|
529
|
+
logger.warning(
|
|
530
|
+
"mailbox tone failed before the recording timer", exc_info=True
|
|
531
|
+
)
|
|
532
|
+
await asyncio.sleep(_VOICEMAIL_RECORD_SECONDS)
|
|
533
|
+
logger.info(
|
|
534
|
+
"mailbox stopped recording after %ss and cut the line",
|
|
535
|
+
_VOICEMAIL_RECORD_SECONDS,
|
|
536
|
+
)
|
|
537
|
+
self._end_requested.set()
|
|
538
|
+
except asyncio.CancelledError:
|
|
539
|
+
raise
|
|
540
|
+
except Exception:
|
|
541
|
+
# A mailbox that fails to hang up leaves the watchdog to end the call, which is the
|
|
542
|
+
# behaviour this replaces rather than a new failure.
|
|
543
|
+
logger.warning("mailbox recording timer failed", exc_info=True)
|
|
544
|
+
|
|
545
|
+
async def _play_voicemail_tone(self, style: str, session: "AgentSession") -> None:
|
|
546
|
+
"""Play the tone a mailbox plays once its greeting has finished.
|
|
547
|
+
|
|
548
|
+
The greeting comes either from a catalogue recording or from this session speaking the
|
|
549
|
+
persona's opening line. The tone is the part neither can carry, and without it a greeting that
|
|
550
|
+
says "leave a message after the tone" asks the agent to wait for something that never comes.
|
|
551
|
+
|
|
552
|
+
Generated rather than fetched: a sine burst is a sine burst, and an asset would be a
|
|
553
|
+
download, a licence and a catalogue for something twelve lines of arithmetic produce. Handed
|
|
554
|
+
over eagerly rather than lazily, since the player consumes the iterator inside its mixer
|
|
555
|
+
task and a failure there can take the ambience down with it.
|
|
556
|
+
"""
|
|
557
|
+
shape = _VOICEMAIL_TONE_BY_STYLE.get(style)
|
|
558
|
+
if shape is None:
|
|
559
|
+
return
|
|
560
|
+
hz, seconds = shape
|
|
561
|
+
try:
|
|
562
|
+
# After the greeting, not at a guessed offset: wait for the mailbox's own first
|
|
563
|
+
# committed turn, which is this session's assistant role, then leave a beat.
|
|
564
|
+
recorded = os.environ.get("HARNESS_VOICEMAIL_CLIP", "").strip()
|
|
565
|
+
if recorded:
|
|
566
|
+
# Awaiting the greeting's own handle is exact where a duration would be a guess.
|
|
567
|
+
if self._voicemail_greeting is not None:
|
|
568
|
+
await self._voicemail_greeting
|
|
569
|
+
else:
|
|
570
|
+
loop = asyncio.get_running_loop()
|
|
571
|
+
deadline = loop.time() + _VOICEMAIL_TONE_WAIT_SECONDS
|
|
572
|
+
while loop.time() < deadline:
|
|
573
|
+
if any(
|
|
574
|
+
message["content"]
|
|
575
|
+
for message in _session_messages(session)
|
|
576
|
+
if message["role"] == "assistant"
|
|
577
|
+
):
|
|
578
|
+
break
|
|
579
|
+
await asyncio.sleep(0.2)
|
|
580
|
+
else:
|
|
581
|
+
logger.warning(
|
|
582
|
+
"mailbox said nothing in %ss; no tone",
|
|
583
|
+
_VOICEMAIL_TONE_WAIT_SECONDS,
|
|
584
|
+
)
|
|
585
|
+
return
|
|
586
|
+
await asyncio.sleep(_VOICEMAIL_TONE_GAP_SECONDS)
|
|
587
|
+
player = getattr(self, "_background_player", None)
|
|
588
|
+
if player is None:
|
|
589
|
+
return
|
|
590
|
+
# A recording that ends with its own tone replaces this one rather than preceding it.
|
|
591
|
+
# Two beeps is worse than one, and the catalogue records which clips carry theirs.
|
|
592
|
+
if os.environ.get("HARNESS_VOICEMAIL_CLIP_HAS_TONE", "").strip() == "1":
|
|
593
|
+
return
|
|
594
|
+
spoken = [_tone_frame(hz, seconds)]
|
|
595
|
+
|
|
596
|
+
async def frames() -> Any:
|
|
597
|
+
for frame in spoken:
|
|
598
|
+
yield frame
|
|
599
|
+
|
|
600
|
+
player.play(AudioConfig(frames(), volume=_VOICEMAIL_TONE_VOLUME))
|
|
601
|
+
except asyncio.CancelledError:
|
|
602
|
+
raise
|
|
603
|
+
except Exception:
|
|
604
|
+
# A mailbox without its tone is a weaker test, never a failed call.
|
|
605
|
+
logger.warning("voicemail tone not played", exc_info=True)
|
|
606
|
+
|
|
607
|
+
async def _stop_background_audio(self) -> None:
|
|
608
|
+
"""Close the ambience player and remove any clip downloaded for it.
|
|
609
|
+
|
|
610
|
+
Without this the mixer task, its audio source and the published track outlive the call,
|
|
611
|
+
and a suite leaks one of each (plus a temp file) per scenario.
|
|
612
|
+
"""
|
|
613
|
+
for name in ("_voicemail_tone_task", "_mailbox_close_task"):
|
|
614
|
+
pending = getattr(self, name, None)
|
|
615
|
+
if pending is not None:
|
|
616
|
+
setattr(self, name, None)
|
|
617
|
+
if not pending.done():
|
|
618
|
+
pending.cancel()
|
|
619
|
+
player = getattr(self, "_background_player", None)
|
|
620
|
+
if player is not None:
|
|
621
|
+
self._background_player = None
|
|
622
|
+
try:
|
|
623
|
+
await player.aclose()
|
|
624
|
+
except Exception:
|
|
625
|
+
logger.warning("background audio not closed cleanly", exc_info=True)
|
|
626
|
+
downloaded = getattr(self, "_background_noise_file", None)
|
|
627
|
+
if downloaded:
|
|
628
|
+
self._background_noise_file = None
|
|
629
|
+
try:
|
|
630
|
+
Path(downloaded).unlink(missing_ok=True)
|
|
631
|
+
except OSError:
|
|
632
|
+
logger.warning("background audio clip not removed: %s", downloaded)
|
|
633
|
+
|
|
634
|
+
def open_conversation(self) -> None:
|
|
635
|
+
if self._session is None:
|
|
636
|
+
raise RuntimeError("simulator_session_not_started")
|
|
637
|
+
if self._voicemail_greeting is not None:
|
|
638
|
+
# A recording has already greeted, and a mailbox does not greet twice: a spoken line on
|
|
639
|
+
# top of the clip is one mailbox answering in two voices.
|
|
640
|
+
return
|
|
641
|
+
initial_message = self._persona.persona.get("initial_message")
|
|
642
|
+
if isinstance(initial_message, str) and initial_message.strip():
|
|
643
|
+
self._session.say(initial_message.strip())
|
|
644
|
+
return
|
|
645
|
+
self._session.generate_reply()
|
|
646
|
+
|
|
647
|
+
_mailbox_greeted: bool = False
|
|
648
|
+
|
|
649
|
+
async def llm_node(self, chat_ctx, tools, model_settings):
|
|
650
|
+
"""A mailbox speaks once and then never again, counted here rather than asked of the model.
|
|
651
|
+
|
|
652
|
+
One turn is allowed, not none, because a mailbox without a recording greets through this path.
|
|
653
|
+
Where a recording has already greeted, no turn is allowed at all.
|
|
654
|
+
"""
|
|
655
|
+
if _answered_by_voicemail():
|
|
656
|
+
if self._mailbox_greeted or self._voicemail_greeting is not None:
|
|
657
|
+
return
|
|
658
|
+
self._mailbox_greeted = True
|
|
659
|
+
async for chunk in super().llm_node(chat_ctx, tools, model_settings):
|
|
660
|
+
yield chunk
|
|
661
|
+
|
|
662
|
+
async def transcription_node(
|
|
663
|
+
self,
|
|
664
|
+
text: AsyncIterable[str | TimedString],
|
|
665
|
+
model_settings: ModelSettings,
|
|
666
|
+
):
|
|
667
|
+
async for chunk in text:
|
|
668
|
+
logger.debug(
|
|
669
|
+
"Simulator transcription chunk",
|
|
670
|
+
extra={"timed": isinstance(chunk, TimedString)},
|
|
671
|
+
)
|
|
672
|
+
yield chunk
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
class LiveKitEngine(BaseEngine):
|
|
676
|
+
async def run(
|
|
677
|
+
self,
|
|
678
|
+
agent_definition: AgentDefinition | None = None,
|
|
679
|
+
livekit_runtime: LiveKitSimulatorRuntime | None = None,
|
|
680
|
+
scenario: Scenario | None = None,
|
|
681
|
+
simulator: SimulatorAgentDefinition | None = None,
|
|
682
|
+
num_scenarios: int = 1,
|
|
683
|
+
topic: str | None = None,
|
|
684
|
+
record_audio: bool = False,
|
|
685
|
+
recorder_sample_rate: int = 8000,
|
|
686
|
+
recorder_join_delay: float = 0.2,
|
|
687
|
+
min_turn_messages: int = 8,
|
|
688
|
+
max_seconds: float = 45.0,
|
|
689
|
+
connect_timeout: float = 15.0,
|
|
690
|
+
readiness_timeout: float = 30.0,
|
|
691
|
+
cleanup_timeout: float = 30.0,
|
|
692
|
+
conversation_direction: str = "simulator_first",
|
|
693
|
+
agent_first_silence_timeout_seconds: float = 120.0,
|
|
694
|
+
recording_root: str | Path = "recordings",
|
|
695
|
+
recording_case_directory: str | Path | None = None,
|
|
696
|
+
run_id: str | None = None,
|
|
697
|
+
max_concurrency: int = 1,
|
|
698
|
+
on_case_complete: Callable[[int, TestCaseResult], Awaitable[None]]
|
|
699
|
+
| None = None,
|
|
700
|
+
on_case_start: Callable[[int], Awaitable[None]] | None = None,
|
|
701
|
+
**kwargs,
|
|
702
|
+
) -> TestReport:
|
|
703
|
+
if agent_definition is None:
|
|
704
|
+
raise ValueError("LiveKitEngine requires 'agent_definition'.")
|
|
705
|
+
runtime = _resolve_livekit_runtime(agent_definition, livekit_runtime)
|
|
706
|
+
if conversation_direction not in {"simulator_first", "agent_first"}:
|
|
707
|
+
raise ValueError(
|
|
708
|
+
"conversation_direction must be simulator_first or agent_first"
|
|
709
|
+
)
|
|
710
|
+
if agent_first_silence_timeout_seconds <= 0:
|
|
711
|
+
raise ValueError("agent_first_silence_timeout_seconds must be positive")
|
|
712
|
+
if scenario is None:
|
|
713
|
+
generator = ScenarioGenerator(
|
|
714
|
+
agent_definition,
|
|
715
|
+
llm_config=(
|
|
716
|
+
simulator.llm
|
|
717
|
+
if simulator is not None
|
|
718
|
+
else _default_simulator_llm_config()
|
|
719
|
+
),
|
|
720
|
+
)
|
|
721
|
+
if topic is None:
|
|
722
|
+
simulator_context = (
|
|
723
|
+
simulator.instructions
|
|
724
|
+
if simulator and simulator.instructions
|
|
725
|
+
else ""
|
|
726
|
+
)
|
|
727
|
+
topic = (
|
|
728
|
+
simulator_context
|
|
729
|
+
or agent_definition.system_prompt
|
|
730
|
+
or "customer support scenarios"
|
|
731
|
+
).strip()
|
|
732
|
+
personas = await generator.generate(
|
|
733
|
+
topic=topic,
|
|
734
|
+
num_personas=num_scenarios,
|
|
735
|
+
)
|
|
736
|
+
scenario = Scenario(name="Generated Scenario", dataset=personas)
|
|
737
|
+
transport = agent_definition.transport or TelephonyTransport()
|
|
738
|
+
profile = _resolve_target_profile(transport.kind)
|
|
739
|
+
# Computed before the room-name checks below: a serial (concurrency-1)
|
|
740
|
+
# run is what makes reusing one fixed room across cases safe, so the
|
|
741
|
+
# verbatim check needs this value, not just the dataset size.
|
|
742
|
+
# Cases run concurrently up to ``max_concurrency`` (bounded by the
|
|
743
|
+
# customer agent's own session capacity). SIP legs stay serial: a run
|
|
744
|
+
# leases a single DID, so overlapping calls would collide. Ask the
|
|
745
|
+
# resolved profile rather than re-branching on ``transport.kind``.
|
|
746
|
+
case_concurrency = (
|
|
747
|
+
1
|
|
748
|
+
if profile.is_sip
|
|
749
|
+
else max(
|
|
750
|
+
1,
|
|
751
|
+
min(
|
|
752
|
+
int(max_concurrency or 1),
|
|
753
|
+
_voice_max_case_concurrency(),
|
|
754
|
+
len(scenario.dataset),
|
|
755
|
+
),
|
|
756
|
+
)
|
|
757
|
+
)
|
|
758
|
+
if (
|
|
759
|
+
runtime.room_name_verbatim
|
|
760
|
+
and len(scenario.dataset) != 1
|
|
761
|
+
and case_concurrency != 1
|
|
762
|
+
):
|
|
763
|
+
raise ValueError(
|
|
764
|
+
"room_name_verbatim_requires_serial_cases: a fixed room hosts "
|
|
765
|
+
"one case at a time; run with max_concurrency=1"
|
|
766
|
+
)
|
|
767
|
+
if (
|
|
768
|
+
runtime.room_mode == "external"
|
|
769
|
+
and len(scenario.dataset) > 1
|
|
770
|
+
and not _has_room_template(runtime.room_name)
|
|
771
|
+
):
|
|
772
|
+
raise ValueError(
|
|
773
|
+
"external_room_template_required: concurrent-safe multi-case runs "
|
|
774
|
+
"need {run_id}, {test_case_id}, or {index} in room_name"
|
|
775
|
+
)
|
|
776
|
+
if not profile.uses_external_room and runtime.room_mode != "managed":
|
|
777
|
+
raise ValueError("managed_transport_requires_managed_room")
|
|
778
|
+
if (
|
|
779
|
+
profile.receives_inbound_call
|
|
780
|
+
and len(scenario.dataset) > 1
|
|
781
|
+
and not _has_room_template(runtime.room_name)
|
|
782
|
+
and not runtime.room_name_verbatim
|
|
783
|
+
):
|
|
784
|
+
raise ValueError(
|
|
785
|
+
"sip_inbound_room_template_required: multi-case inbound runs "
|
|
786
|
+
"need {run_id} or {test_case_id} in room_name"
|
|
787
|
+
)
|
|
788
|
+
cleanup_timeout = min(cleanup_timeout, _MAX_CLEANUP_TIMEOUT_SECONDS)
|
|
789
|
+
current_run_id = run_id or new_run_id()
|
|
790
|
+
if recording_case_directory is not None and len(scenario.dataset) != 1:
|
|
791
|
+
raise ValueError(
|
|
792
|
+
"recording_case_directory requires a single-persona scenario"
|
|
793
|
+
)
|
|
794
|
+
invocation_id = uuid4().hex[:12]
|
|
795
|
+
report = TestReport()
|
|
796
|
+
|
|
797
|
+
case_semaphore = asyncio.Semaphore(case_concurrency)
|
|
798
|
+
|
|
799
|
+
async def _run_case(index: int, persona: Persona) -> TestCaseResult:
|
|
800
|
+
async with case_semaphore:
|
|
801
|
+
# Mark this case's row ONGOING the moment it claims a concurrency
|
|
802
|
+
# slot — cases still queued behind the semaphore stay PENDING.
|
|
803
|
+
# Best-effort and engine-agnostic; a failed ping never fails the
|
|
804
|
+
# case (the backend gates the update on PENDING).
|
|
805
|
+
if on_case_start is not None:
|
|
806
|
+
try:
|
|
807
|
+
await on_case_start(index)
|
|
808
|
+
except Exception as exc: # noqa: BLE001
|
|
809
|
+
logger.error(
|
|
810
|
+
"voice case start callback failed",
|
|
811
|
+
exc_info=redacted_exc_info(exc),
|
|
812
|
+
extra={
|
|
813
|
+
"run_id": current_run_id,
|
|
814
|
+
"case_index": index,
|
|
815
|
+
},
|
|
816
|
+
)
|
|
817
|
+
persona_ref = persona.version or persona.content_hash()
|
|
818
|
+
test_case_id = derive_test_case_id(
|
|
819
|
+
current_run_id,
|
|
820
|
+
persona_ref,
|
|
821
|
+
index,
|
|
822
|
+
)
|
|
823
|
+
room_name = _resolve_room_name(
|
|
824
|
+
runtime,
|
|
825
|
+
run_id=current_run_id,
|
|
826
|
+
test_case_id=test_case_id,
|
|
827
|
+
index=index,
|
|
828
|
+
invocation_id=invocation_id,
|
|
829
|
+
)
|
|
830
|
+
case_directory = (
|
|
831
|
+
Path(recording_case_directory)
|
|
832
|
+
if recording_case_directory is not None
|
|
833
|
+
else Path(recording_root) / current_run_id / test_case_id
|
|
834
|
+
)
|
|
835
|
+
try:
|
|
836
|
+
outcome = await self._run_single_test_case(
|
|
837
|
+
agent_definition,
|
|
838
|
+
runtime,
|
|
839
|
+
persona,
|
|
840
|
+
simulator,
|
|
841
|
+
run_id=current_run_id,
|
|
842
|
+
test_case_id=test_case_id,
|
|
843
|
+
invocation_id=invocation_id,
|
|
844
|
+
room_name=room_name,
|
|
845
|
+
case_directory=case_directory,
|
|
846
|
+
record_audio=record_audio,
|
|
847
|
+
recorder_sample_rate=recorder_sample_rate,
|
|
848
|
+
recorder_join_delay=recorder_join_delay,
|
|
849
|
+
min_turn_messages=min_turn_messages,
|
|
850
|
+
max_seconds=max_seconds,
|
|
851
|
+
connect_timeout=connect_timeout,
|
|
852
|
+
readiness_timeout=readiness_timeout,
|
|
853
|
+
cleanup_timeout=cleanup_timeout,
|
|
854
|
+
conversation_direction=conversation_direction,
|
|
855
|
+
agent_first_silence_timeout_seconds=agent_first_silence_timeout_seconds,
|
|
856
|
+
)
|
|
857
|
+
except asyncio.CancelledError:
|
|
858
|
+
raise
|
|
859
|
+
except Exception as exc: # noqa: BLE001
|
|
860
|
+
# One case crashing must neither sink the batch nor leave a
|
|
861
|
+
# hole that shifts the positional result-to-CallExecution
|
|
862
|
+
# mapping — emit a dense failed case in its slot.
|
|
863
|
+
logger.error(
|
|
864
|
+
"voice case crashed",
|
|
865
|
+
exc_info=redacted_exc_info(exc),
|
|
866
|
+
extra={
|
|
867
|
+
"run_id": current_run_id,
|
|
868
|
+
"test_case_id": test_case_id,
|
|
869
|
+
},
|
|
870
|
+
)
|
|
871
|
+
failure = SimulationFailure(
|
|
872
|
+
stage=FailureStage.RUNNING,
|
|
873
|
+
code="case_execution_error",
|
|
874
|
+
message=f"{type(exc).__name__}: {exc}",
|
|
875
|
+
retryable=False,
|
|
876
|
+
)
|
|
877
|
+
result = TestCaseResult(
|
|
878
|
+
persona=persona,
|
|
879
|
+
transcript="",
|
|
880
|
+
messages=[],
|
|
881
|
+
metadata={
|
|
882
|
+
"engine": "livekit",
|
|
883
|
+
"run_id": current_run_id,
|
|
884
|
+
"test_case_id": test_case_id,
|
|
885
|
+
"invocation_id": invocation_id,
|
|
886
|
+
"status": TestCaseStatus.FAILED.value,
|
|
887
|
+
"room_name": room_name,
|
|
888
|
+
"room_mode": runtime.room_mode,
|
|
889
|
+
"failure": failure.model_dump(
|
|
890
|
+
mode="json", exclude_none=True
|
|
891
|
+
),
|
|
892
|
+
},
|
|
893
|
+
)
|
|
894
|
+
else:
|
|
895
|
+
metadata = {
|
|
896
|
+
"engine": "livekit",
|
|
897
|
+
"run_id": current_run_id,
|
|
898
|
+
"test_case_id": test_case_id,
|
|
899
|
+
"invocation_id": invocation_id,
|
|
900
|
+
"status": outcome.status.value,
|
|
901
|
+
"room_name": room_name,
|
|
902
|
+
"room_mode": runtime.room_mode,
|
|
903
|
+
**outcome.metadata,
|
|
904
|
+
}
|
|
905
|
+
if outcome.failure is not None:
|
|
906
|
+
metadata["failure"] = outcome.failure.model_dump(
|
|
907
|
+
mode="json",
|
|
908
|
+
exclude_none=True,
|
|
909
|
+
)
|
|
910
|
+
if outcome.evidence:
|
|
911
|
+
metadata["evidence"] = [
|
|
912
|
+
item.model_dump(mode="json", exclude_none=True)
|
|
913
|
+
for item in outcome.evidence
|
|
914
|
+
]
|
|
915
|
+
if outcome.provider_artifacts:
|
|
916
|
+
metadata["provider_artifacts"] = [
|
|
917
|
+
entry.model_dump(mode="json", exclude_none=True)
|
|
918
|
+
for entry in outcome.provider_artifacts
|
|
919
|
+
]
|
|
920
|
+
result = TestCaseResult(
|
|
921
|
+
persona=persona,
|
|
922
|
+
transcript=outcome.transcript,
|
|
923
|
+
messages=outcome.messages,
|
|
924
|
+
metadata=metadata,
|
|
925
|
+
audio_input_path=outcome.audio_input_path,
|
|
926
|
+
audio_output_path=outcome.audio_output_path,
|
|
927
|
+
audio_combined_path=outcome.audio_combined_path,
|
|
928
|
+
audio_stereo_path=outcome.audio_stereo_path,
|
|
929
|
+
)
|
|
930
|
+
|
|
931
|
+
# Stream the finished case AFTER releasing the semaphore — a slow
|
|
932
|
+
# result PATCH (recording upload) must not hold a concurrency slot.
|
|
933
|
+
# A streaming error never fails the case; finalize reconciles it.
|
|
934
|
+
if on_case_complete is not None:
|
|
935
|
+
try:
|
|
936
|
+
await on_case_complete(index, result)
|
|
937
|
+
except Exception as exc: # noqa: BLE001
|
|
938
|
+
logger.error(
|
|
939
|
+
"voice case stream callback failed",
|
|
940
|
+
exc_info=redacted_exc_info(exc),
|
|
941
|
+
extra={
|
|
942
|
+
"run_id": current_run_id,
|
|
943
|
+
"test_case_id": test_case_id,
|
|
944
|
+
},
|
|
945
|
+
)
|
|
946
|
+
return result
|
|
947
|
+
|
|
948
|
+
# ``gather`` preserves argument order regardless of completion order, so
|
|
949
|
+
# ``report.results`` stays in dataset order — the positional contract the
|
|
950
|
+
# FutureAGI sink relies on to map results to pre-allocated CallExecutions.
|
|
951
|
+
results = await asyncio.gather(
|
|
952
|
+
*(
|
|
953
|
+
_run_case(index, persona)
|
|
954
|
+
for index, persona in enumerate(scenario.dataset)
|
|
955
|
+
)
|
|
956
|
+
)
|
|
957
|
+
report.results.extend(results)
|
|
958
|
+
return report
|
|
959
|
+
|
|
960
|
+
async def _run_single_test_case(
|
|
961
|
+
self,
|
|
962
|
+
agent_definition: AgentDefinition,
|
|
963
|
+
runtime: LiveKitSimulatorRuntime,
|
|
964
|
+
persona: Persona,
|
|
965
|
+
simulator: SimulatorAgentDefinition | None,
|
|
966
|
+
*,
|
|
967
|
+
run_id: str,
|
|
968
|
+
test_case_id: str,
|
|
969
|
+
invocation_id: str,
|
|
970
|
+
room_name: str,
|
|
971
|
+
case_directory: Path,
|
|
972
|
+
record_audio: bool,
|
|
973
|
+
recorder_sample_rate: int,
|
|
974
|
+
recorder_join_delay: float,
|
|
975
|
+
min_turn_messages: int,
|
|
976
|
+
max_seconds: float,
|
|
977
|
+
connect_timeout: float,
|
|
978
|
+
readiness_timeout: float,
|
|
979
|
+
cleanup_timeout: float,
|
|
980
|
+
conversation_direction: str,
|
|
981
|
+
agent_first_silence_timeout_seconds: float,
|
|
982
|
+
) -> _CaseOutcome:
|
|
983
|
+
# Teardown is a run of independent steps that each used to take the full
|
|
984
|
+
# ``cleanup_timeout``. Ten of them at up to sixty seconds is six hundred seconds of
|
|
985
|
+
# cleanup against a run budget of five hundred and seventy, so one slow teardown spent
|
|
986
|
+
# the whole budget and the case was discarded as a timeout with its conversation already
|
|
987
|
+
# finished. Share one deadline across the run instead, started at the first cleanup that
|
|
988
|
+
# actually waits, so every path gets the same bound however it got there.
|
|
989
|
+
_cleanup_started: list[float] = []
|
|
990
|
+
|
|
991
|
+
def _cleanup_budget(
|
|
992
|
+
cap: float = _CLEANUP_STEP_TIMEOUT_SECONDS,
|
|
993
|
+
) -> float:
|
|
994
|
+
if not _cleanup_started:
|
|
995
|
+
_cleanup_started.append(time.monotonic())
|
|
996
|
+
spent = time.monotonic() - _cleanup_started[0]
|
|
997
|
+
remaining = max(0.1, cleanup_timeout - spent)
|
|
998
|
+
return min(cap, remaining)
|
|
999
|
+
|
|
1000
|
+
api_key = os.environ.get(runtime.api_key_env)
|
|
1001
|
+
api_secret = os.environ.get(runtime.api_secret_env)
|
|
1002
|
+
if not api_key or not api_secret:
|
|
1003
|
+
return _failure_outcome(
|
|
1004
|
+
TestCaseStatus.FAILED,
|
|
1005
|
+
FailureStage.PREPARING,
|
|
1006
|
+
"livekit_credentials_missing",
|
|
1007
|
+
f"{runtime.api_key_env} and {runtime.api_secret_env} are required",
|
|
1008
|
+
)
|
|
1009
|
+
simulator_identity = _simulator_participant_identity(persona, test_case_id)
|
|
1010
|
+
recorder_identity = f"fagi-recorder-{test_case_id[-12:]}"
|
|
1011
|
+
room = rtc.Room()
|
|
1012
|
+
models: LiveKitModels | None = None
|
|
1013
|
+
recorder: RoomRecorder | None = None
|
|
1014
|
+
customer_agent: _TestRunnerAgent | None = None
|
|
1015
|
+
session: AgentSession | None = None
|
|
1016
|
+
api_client: api.LiveKitAPI | None = None
|
|
1017
|
+
target: _TargetParticipant | None = None
|
|
1018
|
+
target_transcription_mode = False
|
|
1019
|
+
target_transcription_handler_registered = False
|
|
1020
|
+
target_transcription_tasks: set[asyncio.Task[None]] = set()
|
|
1021
|
+
# Set the moment the conversation ends. A target transcription stream
|
|
1022
|
+
# still in flight at that point is the target's final utterance; it must
|
|
1023
|
+
# be recorded into the transcript, but WITHOUT triggering another
|
|
1024
|
+
# simulator reply (the call is over).
|
|
1025
|
+
conversation_ended = asyncio.Event()
|
|
1026
|
+
# Every target utterance, captured straight off its transcription stream
|
|
1027
|
+
# independently of the simulator session. Once the session starts
|
|
1028
|
+
# draining it rejects new input ("speech scheduling is paused"), so a
|
|
1029
|
+
# target closing delivered after the simulator is done never reaches the
|
|
1030
|
+
# chat context. These are merged into the report so the trailing target
|
|
1031
|
+
# turn is never lost.
|
|
1032
|
+
captured_target_turns: list[dict[str, Any]] = []
|
|
1033
|
+
# agent_first (target greets first): the target can publish its greeting
|
|
1034
|
+
# transcription before the main handler is registered post-readiness, and
|
|
1035
|
+
# the LiveKit client DROPS a text-stream header that arrives with no
|
|
1036
|
+
# handler. So for managed external-room agent_first we register an early
|
|
1037
|
+
# buffer handler right after connect, defer the target dispatch until the
|
|
1038
|
+
# buffer is live, and drain the buffered streams through the (unchanged,
|
|
1039
|
+
# unconditional) main handler once the target is selected.
|
|
1040
|
+
pending_target_transcriptions: list[tuple["rtc.TextStreamReader", str]] = []
|
|
1041
|
+
target_dispatch_deferred = False
|
|
1042
|
+
_MAX_BUFFERED_TARGET_STREAMS = 16
|
|
1043
|
+
managed_room_owned = runtime.room_mode == "managed"
|
|
1044
|
+
room_connected = False
|
|
1045
|
+
cleanup_errors: list[str] = []
|
|
1046
|
+
outcome: _CaseOutcome | None = None
|
|
1047
|
+
sip_dispatch_rule_id: str | None = None
|
|
1048
|
+
sip_dispatch_rule_created = False
|
|
1049
|
+
call_originator: CallOriginator | None = None
|
|
1050
|
+
provider_call_id: str | None = None
|
|
1051
|
+
provider_termination_source: str | None = None
|
|
1052
|
+
reconciled_call_ids: list[str] = []
|
|
1053
|
+
caller_verification: str | None = None
|
|
1054
|
+
audio_bridge: LiveKitAudioBridge | None = None
|
|
1055
|
+
bridge_task: asyncio.Task[None] | None = None
|
|
1056
|
+
case_started_at = datetime.now(timezone.utc)
|
|
1057
|
+
transport = agent_definition.transport or TelephonyTransport()
|
|
1058
|
+
profile = _resolve_target_profile(transport.kind)
|
|
1059
|
+
provider_target = agent_definition.target
|
|
1060
|
+
effective_target_identity = agent_definition.target_participant_identity
|
|
1061
|
+
effective_readiness_timeout = (
|
|
1062
|
+
transport.readiness_timeout_seconds
|
|
1063
|
+
if profile.receives_inbound_call
|
|
1064
|
+
and transport.readiness_timeout_seconds is not None
|
|
1065
|
+
else readiness_timeout
|
|
1066
|
+
)
|
|
1067
|
+
sip_answer_timeout = transport.answer_timeout_seconds or max(
|
|
1068
|
+
connect_timeout, 60.0
|
|
1069
|
+
)
|
|
1070
|
+
|
|
1071
|
+
def buffer_target_transcription(
|
|
1072
|
+
reader: "rtc.TextStreamReader",
|
|
1073
|
+
participant_identity: str,
|
|
1074
|
+
) -> None:
|
|
1075
|
+
# Early (pre-readiness) handler for managed external-room agent_first.
|
|
1076
|
+
# ``target.audio_track_sid`` isn't known yet, so attribute by the same
|
|
1077
|
+
# exclusion ``_wait_for_target_audio`` uses — anything that is not the
|
|
1078
|
+
# simulator or recorder is target-worthy. The main handler re-applies
|
|
1079
|
+
# the strict track/identity filter when draining, so over-buffering is
|
|
1080
|
+
# safe.
|
|
1081
|
+
nonlocal target_transcription_mode
|
|
1082
|
+
pid = str(participant_identity)
|
|
1083
|
+
if pid in (simulator_identity, recorder_identity):
|
|
1084
|
+
return
|
|
1085
|
+
if len(pending_target_transcriptions) >= _MAX_BUFFERED_TARGET_STREAMS:
|
|
1086
|
+
logger.warning(
|
|
1087
|
+
"target_transcription_buffer_full: dropping stream (buffered=%d)",
|
|
1088
|
+
len(pending_target_transcriptions),
|
|
1089
|
+
)
|
|
1090
|
+
return
|
|
1091
|
+
pending_target_transcriptions.append((reader, pid))
|
|
1092
|
+
# Kill the duplicate-response race at the source: the target is
|
|
1093
|
+
# speaking, so disable the simulator's STT now — otherwise STT would
|
|
1094
|
+
# also transcribe the greeting and emit a second, duplicate reply.
|
|
1095
|
+
# Dispatch is deferred until after ``session.start()``, so a buffered
|
|
1096
|
+
# stream implies a live session.
|
|
1097
|
+
if not target_transcription_mode and session is not None:
|
|
1098
|
+
session.input.set_audio_enabled(False)
|
|
1099
|
+
session.clear_user_turn()
|
|
1100
|
+
target_transcription_mode = True
|
|
1101
|
+
|
|
1102
|
+
try:
|
|
1103
|
+
if managed_room_owned:
|
|
1104
|
+
api_client = api.LiveKitAPI(
|
|
1105
|
+
_api_url(str(runtime.url)),
|
|
1106
|
+
api_key,
|
|
1107
|
+
api_secret,
|
|
1108
|
+
)
|
|
1109
|
+
if not profile.places_outbound_call:
|
|
1110
|
+
# The pool's dispatch rule stays live for the whole run, so a
|
|
1111
|
+
# stray inbound call can re-create this room at any moment
|
|
1112
|
+
# between cases — drain it before trusting it as ours.
|
|
1113
|
+
if runtime.room_name_verbatim:
|
|
1114
|
+
try:
|
|
1115
|
+
polls = await asyncio.wait_for(
|
|
1116
|
+
_ensure_room_absent(api_client, room_name),
|
|
1117
|
+
timeout=connect_timeout,
|
|
1118
|
+
)
|
|
1119
|
+
logger.info(
|
|
1120
|
+
"leased room drained",
|
|
1121
|
+
extra={
|
|
1122
|
+
"run_id": run_id,
|
|
1123
|
+
"test_case_id": test_case_id,
|
|
1124
|
+
"room_name": room_name,
|
|
1125
|
+
"polls": polls,
|
|
1126
|
+
},
|
|
1127
|
+
)
|
|
1128
|
+
except asyncio.TimeoutError:
|
|
1129
|
+
outcome = _failure_outcome(
|
|
1130
|
+
TestCaseStatus.TIMED_OUT,
|
|
1131
|
+
FailureStage.PREPARING,
|
|
1132
|
+
"livekit_room_drain_timeout",
|
|
1133
|
+
"The leased simulator room was still occupied when its drain deadline passed",
|
|
1134
|
+
retryable=True,
|
|
1135
|
+
)
|
|
1136
|
+
except Exception as exc: # noqa: BLE001
|
|
1137
|
+
logger.warning(
|
|
1138
|
+
"leased room drain failed",
|
|
1139
|
+
exc_info=redacted_exc_info(exc),
|
|
1140
|
+
extra={
|
|
1141
|
+
"run_id": run_id,
|
|
1142
|
+
"test_case_id": test_case_id,
|
|
1143
|
+
"room_name": room_name,
|
|
1144
|
+
},
|
|
1145
|
+
)
|
|
1146
|
+
outcome = _failure_outcome(
|
|
1147
|
+
TestCaseStatus.FAILED,
|
|
1148
|
+
FailureStage.PREPARING,
|
|
1149
|
+
"livekit_room_drain_failed",
|
|
1150
|
+
"Could not clear the leased simulator room before the call",
|
|
1151
|
+
details=_safe_provider_error_details(
|
|
1152
|
+
exc, operation="room_drain"
|
|
1153
|
+
),
|
|
1154
|
+
)
|
|
1155
|
+
if outcome is None:
|
|
1156
|
+
try:
|
|
1157
|
+
await asyncio.wait_for(
|
|
1158
|
+
api_client.room.create_room(
|
|
1159
|
+
api.CreateRoomRequest(name=room_name)
|
|
1160
|
+
),
|
|
1161
|
+
timeout=connect_timeout,
|
|
1162
|
+
)
|
|
1163
|
+
except asyncio.TimeoutError:
|
|
1164
|
+
outcome = _failure_outcome(
|
|
1165
|
+
TestCaseStatus.TIMED_OUT,
|
|
1166
|
+
FailureStage.PREPARING,
|
|
1167
|
+
"livekit_room_create_timeout",
|
|
1168
|
+
"LiveKit room creation exceeded its deadline",
|
|
1169
|
+
retryable=True,
|
|
1170
|
+
)
|
|
1171
|
+
except Exception as exc:
|
|
1172
|
+
logger.warning(
|
|
1173
|
+
"LiveKit room creation failed",
|
|
1174
|
+
exc_info=redacted_exc_info(exc),
|
|
1175
|
+
extra={
|
|
1176
|
+
"run_id": run_id,
|
|
1177
|
+
"test_case_id": test_case_id,
|
|
1178
|
+
"room_name": room_name,
|
|
1179
|
+
},
|
|
1180
|
+
)
|
|
1181
|
+
outcome = _failure_outcome(
|
|
1182
|
+
TestCaseStatus.FAILED,
|
|
1183
|
+
FailureStage.PREPARING,
|
|
1184
|
+
"livekit_room_create_failed",
|
|
1185
|
+
"Failed to create the LiveKit room",
|
|
1186
|
+
details=_safe_provider_error_details(
|
|
1187
|
+
exc, operation="room_create"
|
|
1188
|
+
),
|
|
1189
|
+
)
|
|
1190
|
+
if outcome is None and profile.uses_external_room:
|
|
1191
|
+
# Defer the target dispatch until AFTER the early buffer
|
|
1192
|
+
# handler is registered and the session is live (both
|
|
1193
|
+
# directions): a native target may greet the moment it
|
|
1194
|
+
# joins even when the simulator is meant to open, and the
|
|
1195
|
+
# LiveKit client drops a text-stream header that arrives
|
|
1196
|
+
# with no handler — the greeting would be lost.
|
|
1197
|
+
target_dispatch_deferred = True
|
|
1198
|
+
elif outcome is None and profile.receives_inbound_call:
|
|
1199
|
+
try:
|
|
1200
|
+
(
|
|
1201
|
+
sip_dispatch_rule_id,
|
|
1202
|
+
sip_dispatch_rule_created,
|
|
1203
|
+
) = await asyncio.wait_for(
|
|
1204
|
+
_ensure_sip_inbound_dispatch(
|
|
1205
|
+
api_client,
|
|
1206
|
+
transport=transport,
|
|
1207
|
+
room_name=room_name,
|
|
1208
|
+
),
|
|
1209
|
+
timeout=connect_timeout,
|
|
1210
|
+
)
|
|
1211
|
+
except asyncio.TimeoutError:
|
|
1212
|
+
outcome = _failure_outcome(
|
|
1213
|
+
TestCaseStatus.TIMED_OUT,
|
|
1214
|
+
FailureStage.PREPARING,
|
|
1215
|
+
"sip_inbound_dispatch_timeout",
|
|
1216
|
+
"SIP inbound dispatch provisioning exceeded its deadline",
|
|
1217
|
+
retryable=True,
|
|
1218
|
+
)
|
|
1219
|
+
except Exception as exc:
|
|
1220
|
+
logger.warning(
|
|
1221
|
+
"SIP inbound dispatch provisioning failed",
|
|
1222
|
+
exc_info=redacted_exc_info(exc),
|
|
1223
|
+
extra={
|
|
1224
|
+
"run_id": run_id,
|
|
1225
|
+
"test_case_id": test_case_id,
|
|
1226
|
+
"room_name": room_name,
|
|
1227
|
+
},
|
|
1228
|
+
)
|
|
1229
|
+
outcome = _failure_outcome(
|
|
1230
|
+
TestCaseStatus.FAILED,
|
|
1231
|
+
FailureStage.PREPARING,
|
|
1232
|
+
"sip_inbound_dispatch_failed",
|
|
1233
|
+
"Failed to provision SIP inbound dispatch",
|
|
1234
|
+
details=_safe_provider_error_details(
|
|
1235
|
+
exc, operation="sip_dispatch"
|
|
1236
|
+
),
|
|
1237
|
+
)
|
|
1238
|
+
if outcome is not None:
|
|
1239
|
+
return outcome
|
|
1240
|
+
# The target resolves who is calling from participant attributes or metadata.
|
|
1241
|
+
# Without the persona's number every scenario looks like the same demo rider and
|
|
1242
|
+
# the agent looks up the wrong account, which reads as an agent bug.
|
|
1243
|
+
caller_phone = str(
|
|
1244
|
+
(persona.persona.get("metadata") or {}).get("caller_phone") or ""
|
|
1245
|
+
).strip()
|
|
1246
|
+
builder = (
|
|
1247
|
+
AccessToken(api_key, api_secret)
|
|
1248
|
+
.with_identity(simulator_identity)
|
|
1249
|
+
.with_grants(VideoGrants(room_join=True, room=room_name))
|
|
1250
|
+
)
|
|
1251
|
+
if caller_phone:
|
|
1252
|
+
builder = builder.with_attributes(
|
|
1253
|
+
{"harness.callerPhone": caller_phone}
|
|
1254
|
+
).with_metadata(json.dumps({"caller_phone": caller_phone}))
|
|
1255
|
+
token = builder.to_jwt()
|
|
1256
|
+
await asyncio.wait_for(
|
|
1257
|
+
room.connect(str(runtime.url), token),
|
|
1258
|
+
timeout=connect_timeout,
|
|
1259
|
+
)
|
|
1260
|
+
room_connected = True
|
|
1261
|
+
if profile.uses_external_room:
|
|
1262
|
+
# Buffer any target greeting that arrives before readiness; the
|
|
1263
|
+
# LiveKit client drops a text-stream header with no handler.
|
|
1264
|
+
room.register_text_stream_handler(
|
|
1265
|
+
TOPIC_TRANSCRIPTION, buffer_target_transcription
|
|
1266
|
+
)
|
|
1267
|
+
target_transcription_handler_registered = True
|
|
1268
|
+
if record_audio:
|
|
1269
|
+
recorder = RoomRecorder(
|
|
1270
|
+
url=str(runtime.url),
|
|
1271
|
+
api_key=api_key,
|
|
1272
|
+
api_secret=api_secret,
|
|
1273
|
+
room_name=room_name,
|
|
1274
|
+
identity=recorder_identity,
|
|
1275
|
+
sample_rate=recorder_sample_rate,
|
|
1276
|
+
output_dir=case_directory / "audio",
|
|
1277
|
+
join_delay_s=recorder_join_delay,
|
|
1278
|
+
)
|
|
1279
|
+
await asyncio.wait_for(
|
|
1280
|
+
recorder.start(),
|
|
1281
|
+
timeout=connect_timeout,
|
|
1282
|
+
)
|
|
1283
|
+
customer_agent, models = await self._create_customer_agent(
|
|
1284
|
+
persona,
|
|
1285
|
+
simulator,
|
|
1286
|
+
# The AGENT's direction; it picks which half of the role block the caller gets.
|
|
1287
|
+
call_type=(
|
|
1288
|
+
"outbound"
|
|
1289
|
+
if os.environ.get("HARNESS_CALL_DIRECTION", "").strip().lower()
|
|
1290
|
+
== "outbound"
|
|
1291
|
+
else "inbound"
|
|
1292
|
+
),
|
|
1293
|
+
# `name` is an identity for dispatch, not a label for the caller to hear.
|
|
1294
|
+
agent_name=agent_definition.description,
|
|
1295
|
+
min_turn_messages=min_turn_messages,
|
|
1296
|
+
)
|
|
1297
|
+
setup = getattr(self, "_last_simulator_setup", {}) or {}
|
|
1298
|
+
_record_simulator_setup(
|
|
1299
|
+
case_directory,
|
|
1300
|
+
persona=persona,
|
|
1301
|
+
instructions=setup.get("instructions", ""),
|
|
1302
|
+
llm_config=setup.get("llm_config"),
|
|
1303
|
+
stt_config=setup.get("stt_config"),
|
|
1304
|
+
tts_config=setup.get("tts_config"),
|
|
1305
|
+
extra={
|
|
1306
|
+
"room_name": room_name,
|
|
1307
|
+
"agent_name": agent_definition.name,
|
|
1308
|
+
"test_case_id": test_case_id,
|
|
1309
|
+
"run_id": run_id,
|
|
1310
|
+
"conversation_direction": conversation_direction,
|
|
1311
|
+
"allow_interruptions": setup.get("allow_interruptions"),
|
|
1312
|
+
"min_endpointing_delay": setup.get("min_endpointing_delay"),
|
|
1313
|
+
"max_endpointing_delay": setup.get("max_endpointing_delay"),
|
|
1314
|
+
"use_tts_aligned_transcript": setup.get(
|
|
1315
|
+
"use_tts_aligned_transcript"
|
|
1316
|
+
),
|
|
1317
|
+
},
|
|
1318
|
+
)
|
|
1319
|
+
sip_participant_identity: str | None = None
|
|
1320
|
+
bridge_identity: str | None = None
|
|
1321
|
+
if profile.places_outbound_call:
|
|
1322
|
+
identity_template = (
|
|
1323
|
+
transport.participant_identity
|
|
1324
|
+
or "sip-caller-{invocation_id}-{test_case_id}"
|
|
1325
|
+
)
|
|
1326
|
+
sip_participant_identity = identity_template.format(
|
|
1327
|
+
test_case_id=test_case_id,
|
|
1328
|
+
run_id=run_id,
|
|
1329
|
+
invocation_id=invocation_id,
|
|
1330
|
+
)
|
|
1331
|
+
if effective_target_identity is None:
|
|
1332
|
+
effective_target_identity = sip_participant_identity
|
|
1333
|
+
elif profile.uses_web_audio_bridge:
|
|
1334
|
+
bridge_identity = (
|
|
1335
|
+
f"fagi-{profile.bridge_provider}-bridge-{test_case_id[-12:]}"
|
|
1336
|
+
)
|
|
1337
|
+
effective_target_identity = bridge_identity
|
|
1338
|
+
session_participant_kinds = None
|
|
1339
|
+
session_participant_identity: str | None = None
|
|
1340
|
+
if profile.joins_as_sip_participant:
|
|
1341
|
+
session_participant_kinds = [rtc.ParticipantKind.PARTICIPANT_KIND_SIP]
|
|
1342
|
+
session_participant_identity = (
|
|
1343
|
+
effective_target_identity or sip_participant_identity
|
|
1344
|
+
)
|
|
1345
|
+
session = await asyncio.wait_for(
|
|
1346
|
+
customer_agent.start_session(
|
|
1347
|
+
room,
|
|
1348
|
+
participant_kinds=session_participant_kinds,
|
|
1349
|
+
participant_identity=session_participant_identity,
|
|
1350
|
+
),
|
|
1351
|
+
timeout=connect_timeout,
|
|
1352
|
+
)
|
|
1353
|
+
if target_dispatch_deferred:
|
|
1354
|
+
# Session + early buffer handler are live; now dispatch the target
|
|
1355
|
+
# so its greeting stream is captured, not dropped.
|
|
1356
|
+
assert api_client is not None
|
|
1357
|
+
dispatch_agent_name = (
|
|
1358
|
+
agent_definition.agent_name or agent_definition.name
|
|
1359
|
+
)
|
|
1360
|
+
try:
|
|
1361
|
+
await asyncio.wait_for(
|
|
1362
|
+
api_client.agent_dispatch.create_dispatch(
|
|
1363
|
+
api.CreateAgentDispatchRequest(
|
|
1364
|
+
agent_name=dispatch_agent_name,
|
|
1365
|
+
room=room_name,
|
|
1366
|
+
metadata=_dispatch_metadata_json(agent_definition),
|
|
1367
|
+
)
|
|
1368
|
+
),
|
|
1369
|
+
timeout=connect_timeout,
|
|
1370
|
+
)
|
|
1371
|
+
except asyncio.TimeoutError:
|
|
1372
|
+
outcome = _failure_outcome(
|
|
1373
|
+
TestCaseStatus.TIMED_OUT,
|
|
1374
|
+
FailureStage.PREPARING,
|
|
1375
|
+
"livekit_dispatch_timeout",
|
|
1376
|
+
"Target agent dispatch exceeded its deadline",
|
|
1377
|
+
retryable=True,
|
|
1378
|
+
)
|
|
1379
|
+
return outcome
|
|
1380
|
+
except Exception as exc:
|
|
1381
|
+
logger.warning(
|
|
1382
|
+
"LiveKit target dispatch failed",
|
|
1383
|
+
exc_info=redacted_exc_info(exc),
|
|
1384
|
+
extra={
|
|
1385
|
+
"run_id": run_id,
|
|
1386
|
+
"test_case_id": test_case_id,
|
|
1387
|
+
"room_name": room_name,
|
|
1388
|
+
},
|
|
1389
|
+
)
|
|
1390
|
+
outcome = _failure_outcome(
|
|
1391
|
+
TestCaseStatus.FAILED,
|
|
1392
|
+
FailureStage.PREPARING,
|
|
1393
|
+
"livekit_dispatch_failed",
|
|
1394
|
+
"Failed to dispatch the target agent",
|
|
1395
|
+
details=_safe_provider_error_details(
|
|
1396
|
+
exc, operation="agent_dispatch"
|
|
1397
|
+
),
|
|
1398
|
+
)
|
|
1399
|
+
return outcome
|
|
1400
|
+
logger.info(
|
|
1401
|
+
"livekit_target_dispatched agent=%s room=%s run=%s case=%s",
|
|
1402
|
+
dispatch_agent_name,
|
|
1403
|
+
room_name,
|
|
1404
|
+
run_id,
|
|
1405
|
+
test_case_id,
|
|
1406
|
+
)
|
|
1407
|
+
if profile.uses_web_audio_bridge:
|
|
1408
|
+
try:
|
|
1409
|
+
connector = profile.build_connector(
|
|
1410
|
+
provider_target,
|
|
1411
|
+
conversation_direction=conversation_direction,
|
|
1412
|
+
)
|
|
1413
|
+
audio_bridge = LiveKitAudioBridge(
|
|
1414
|
+
url=str(runtime.url),
|
|
1415
|
+
api_key=api_key,
|
|
1416
|
+
api_secret=api_secret,
|
|
1417
|
+
room_name=room_name,
|
|
1418
|
+
identity=bridge_identity or "fagi-provider-bridge",
|
|
1419
|
+
connector=connector,
|
|
1420
|
+
)
|
|
1421
|
+
await asyncio.wait_for(
|
|
1422
|
+
audio_bridge.connect(), timeout=connect_timeout
|
|
1423
|
+
)
|
|
1424
|
+
provider_call_id = audio_bridge.call_id
|
|
1425
|
+
bridge_task = asyncio.create_task(audio_bridge.run())
|
|
1426
|
+
except asyncio.TimeoutError:
|
|
1427
|
+
outcome = _failure_outcome(
|
|
1428
|
+
TestCaseStatus.TIMED_OUT,
|
|
1429
|
+
FailureStage.PREPARING,
|
|
1430
|
+
"web_bridge_start_timeout",
|
|
1431
|
+
"Provider web call creation exceeded its deadline",
|
|
1432
|
+
retryable=True,
|
|
1433
|
+
)
|
|
1434
|
+
return outcome
|
|
1435
|
+
except Exception as exc:
|
|
1436
|
+
logger.warning(
|
|
1437
|
+
"Provider web bridge creation failed",
|
|
1438
|
+
exc_info=redacted_exc_info(exc),
|
|
1439
|
+
extra={
|
|
1440
|
+
"run_id": run_id,
|
|
1441
|
+
"test_case_id": test_case_id,
|
|
1442
|
+
"transport": transport.kind,
|
|
1443
|
+
},
|
|
1444
|
+
)
|
|
1445
|
+
outcome = _failure_outcome(
|
|
1446
|
+
TestCaseStatus.FAILED,
|
|
1447
|
+
FailureStage.PREPARING,
|
|
1448
|
+
"web_bridge_start_failed",
|
|
1449
|
+
"Failed to start the provider web bridge",
|
|
1450
|
+
details=_safe_provider_error_details(
|
|
1451
|
+
exc, operation="web_bridge_start"
|
|
1452
|
+
),
|
|
1453
|
+
)
|
|
1454
|
+
return outcome
|
|
1455
|
+
if profile.places_outbound_call and api_client is not None:
|
|
1456
|
+
try:
|
|
1457
|
+
logger.info(
|
|
1458
|
+
"sip_outbound_dialing",
|
|
1459
|
+
extra={
|
|
1460
|
+
"run_id": run_id,
|
|
1461
|
+
"test_case_id": test_case_id,
|
|
1462
|
+
"room_name": room_name,
|
|
1463
|
+
},
|
|
1464
|
+
)
|
|
1465
|
+
await asyncio.wait_for(
|
|
1466
|
+
api_client.sip.create_sip_participant(
|
|
1467
|
+
api.CreateSIPParticipantRequest(
|
|
1468
|
+
sip_trunk_id=transport.sip_trunk_id,
|
|
1469
|
+
sip_number=transport.sip_number,
|
|
1470
|
+
sip_call_to=transport.sip_call_to,
|
|
1471
|
+
room_name=room_name,
|
|
1472
|
+
participant_identity=sip_participant_identity,
|
|
1473
|
+
wait_until_answered=True,
|
|
1474
|
+
play_ringtone=True,
|
|
1475
|
+
)
|
|
1476
|
+
),
|
|
1477
|
+
timeout=sip_answer_timeout,
|
|
1478
|
+
)
|
|
1479
|
+
except asyncio.TimeoutError:
|
|
1480
|
+
outcome = _failure_outcome(
|
|
1481
|
+
TestCaseStatus.TIMED_OUT,
|
|
1482
|
+
FailureStage.PREPARING,
|
|
1483
|
+
"sip_answer_timeout",
|
|
1484
|
+
"Outbound SIP call was not answered before the deadline",
|
|
1485
|
+
retryable=True,
|
|
1486
|
+
)
|
|
1487
|
+
return outcome
|
|
1488
|
+
except Exception as exc:
|
|
1489
|
+
logger.warning(
|
|
1490
|
+
"SIP dial failed",
|
|
1491
|
+
exc_info=redacted_exc_info(exc),
|
|
1492
|
+
extra={
|
|
1493
|
+
"run_id": run_id,
|
|
1494
|
+
"test_case_id": test_case_id,
|
|
1495
|
+
"room_name": room_name,
|
|
1496
|
+
},
|
|
1497
|
+
)
|
|
1498
|
+
outcome = _failure_outcome(
|
|
1499
|
+
TestCaseStatus.FAILED,
|
|
1500
|
+
FailureStage.PREPARING,
|
|
1501
|
+
"sip_dial_failed",
|
|
1502
|
+
"Failed to dial the SIP participant",
|
|
1503
|
+
details=_safe_provider_error_details(exc, operation="sip_dial"),
|
|
1504
|
+
)
|
|
1505
|
+
return outcome
|
|
1506
|
+
if profile.receives_inbound_call:
|
|
1507
|
+
logger.info(
|
|
1508
|
+
"sip_inbound_ready",
|
|
1509
|
+
extra={
|
|
1510
|
+
"run_id": run_id,
|
|
1511
|
+
"test_case_id": test_case_id,
|
|
1512
|
+
"room_name": room_name,
|
|
1513
|
+
"sip_dispatch_rule_id": sip_dispatch_rule_id,
|
|
1514
|
+
"sip_dispatch_rule_created": sip_dispatch_rule_created,
|
|
1515
|
+
},
|
|
1516
|
+
)
|
|
1517
|
+
if runtime.room_name_verbatim and profile.receives_inbound_call:
|
|
1518
|
+
unexpected = _unexpected_participants(
|
|
1519
|
+
room,
|
|
1520
|
+
simulator_identity=simulator_identity,
|
|
1521
|
+
recorder_identity=recorder_identity,
|
|
1522
|
+
)
|
|
1523
|
+
if unexpected:
|
|
1524
|
+
logger.warning(
|
|
1525
|
+
"leased room occupied before dial",
|
|
1526
|
+
extra={
|
|
1527
|
+
"run_id": run_id,
|
|
1528
|
+
"test_case_id": test_case_id,
|
|
1529
|
+
"room_name": room_name,
|
|
1530
|
+
"unexpected_participants": len(unexpected),
|
|
1531
|
+
},
|
|
1532
|
+
)
|
|
1533
|
+
outcome = _failure_outcome(
|
|
1534
|
+
TestCaseStatus.FAILED,
|
|
1535
|
+
FailureStage.PREPARING,
|
|
1536
|
+
"livekit_room_occupied",
|
|
1537
|
+
"Another participant was already in the leased simulator room before the call was placed",
|
|
1538
|
+
retryable=True,
|
|
1539
|
+
)
|
|
1540
|
+
return outcome
|
|
1541
|
+
if transport.inbound_call_originator is not None:
|
|
1542
|
+
name = transport.inbound_call_originator
|
|
1543
|
+
try:
|
|
1544
|
+
call_originator = build_call_originator(transport)
|
|
1545
|
+
originated_call = await asyncio.wait_for(
|
|
1546
|
+
call_originator.start(), timeout=connect_timeout
|
|
1547
|
+
)
|
|
1548
|
+
provider_call_id = originated_call.call_id
|
|
1549
|
+
except asyncio.TimeoutError:
|
|
1550
|
+
outcome = _failure_outcome(
|
|
1551
|
+
TestCaseStatus.TIMED_OUT,
|
|
1552
|
+
FailureStage.PREPARING,
|
|
1553
|
+
f"{name}_call_start_timeout",
|
|
1554
|
+
f"{name.capitalize()} call creation exceeded its deadline",
|
|
1555
|
+
retryable=True,
|
|
1556
|
+
)
|
|
1557
|
+
return outcome
|
|
1558
|
+
except Exception as exc:
|
|
1559
|
+
logger.warning(
|
|
1560
|
+
f"{name.capitalize()} call creation failed",
|
|
1561
|
+
exc_info=redacted_exc_info(exc),
|
|
1562
|
+
extra={
|
|
1563
|
+
"run_id": run_id,
|
|
1564
|
+
"test_case_id": test_case_id,
|
|
1565
|
+
},
|
|
1566
|
+
)
|
|
1567
|
+
outcome = _failure_outcome(
|
|
1568
|
+
TestCaseStatus.FAILED,
|
|
1569
|
+
FailureStage.PREPARING,
|
|
1570
|
+
f"{name}_call_start_failed",
|
|
1571
|
+
f"Failed to start the {name.capitalize()} call",
|
|
1572
|
+
details=_safe_provider_error_details(
|
|
1573
|
+
exc, operation=f"{name}_call_start"
|
|
1574
|
+
),
|
|
1575
|
+
)
|
|
1576
|
+
return outcome
|
|
1577
|
+
target = await _wait_for_target_audio(
|
|
1578
|
+
room,
|
|
1579
|
+
excluded_identities={simulator_identity, recorder_identity},
|
|
1580
|
+
target_identity=effective_target_identity,
|
|
1581
|
+
timeout=effective_readiness_timeout,
|
|
1582
|
+
)
|
|
1583
|
+
if (
|
|
1584
|
+
runtime.room_name_verbatim
|
|
1585
|
+
and profile.receives_inbound_call
|
|
1586
|
+
and transport.originator_from_number
|
|
1587
|
+
):
|
|
1588
|
+
verdict = _caller_matches(
|
|
1589
|
+
target.attributes, transport.originator_from_number
|
|
1590
|
+
)
|
|
1591
|
+
if verdict is False:
|
|
1592
|
+
logger.warning(
|
|
1593
|
+
"leased room wrong caller",
|
|
1594
|
+
extra={
|
|
1595
|
+
"run_id": run_id,
|
|
1596
|
+
"test_case_id": test_case_id,
|
|
1597
|
+
"expected_digits": len(
|
|
1598
|
+
_number_digits(transport.originator_from_number)
|
|
1599
|
+
),
|
|
1600
|
+
"observed_digits": len(
|
|
1601
|
+
_number_digits(
|
|
1602
|
+
target.attributes.get(
|
|
1603
|
+
_SIP_REMOTE_NUMBER_ATTRIBUTE, ""
|
|
1604
|
+
)
|
|
1605
|
+
)
|
|
1606
|
+
),
|
|
1607
|
+
},
|
|
1608
|
+
)
|
|
1609
|
+
caller_verification = "mismatch"
|
|
1610
|
+
raise _LeasedRoomCallerMismatch()
|
|
1611
|
+
elif verdict is None:
|
|
1612
|
+
logger.warning(
|
|
1613
|
+
"leased room caller unverified",
|
|
1614
|
+
extra={"run_id": run_id, "test_case_id": test_case_id},
|
|
1615
|
+
)
|
|
1616
|
+
caller_verification = "unverified"
|
|
1617
|
+
else:
|
|
1618
|
+
caller_verification = "matched"
|
|
1619
|
+
logger.info(
|
|
1620
|
+
"livekit_target_joined identity=%s sid=%s track=%s run=%s case=%s",
|
|
1621
|
+
target.identity,
|
|
1622
|
+
target.sid,
|
|
1623
|
+
target.audio_track_sid,
|
|
1624
|
+
run_id,
|
|
1625
|
+
test_case_id,
|
|
1626
|
+
)
|
|
1627
|
+
# RoomIO auto-links to the first participant that joined — the
|
|
1628
|
+
# recorder, which publishes no audio — so the simulator's STT never
|
|
1629
|
+
# hears the target. Re-point it at the target readiness selected.
|
|
1630
|
+
target_room_io = getattr(session, "room_io", None)
|
|
1631
|
+
if target_room_io is not None:
|
|
1632
|
+
target_room_io.set_participant(target.identity)
|
|
1633
|
+
|
|
1634
|
+
# Swap the early buffer handler for the authoritative one. The
|
|
1635
|
+
# unregister → def → register sequence has NO await between the
|
|
1636
|
+
# unregister and register, so no header can land unhandled in the gap
|
|
1637
|
+
# (LiveKit allows one handler per topic and raises on double-register).
|
|
1638
|
+
if target_transcription_handler_registered:
|
|
1639
|
+
room.unregister_text_stream_handler(TOPIC_TRANSCRIPTION)
|
|
1640
|
+
target_transcription_handler_registered = False
|
|
1641
|
+
|
|
1642
|
+
# The target's mic track stays open for the whole call, so STT never
|
|
1643
|
+
# finalizes a turn; the agent's authoritative turns arrive on its
|
|
1644
|
+
# lk.transcription stream instead. Consume that stream and feed each
|
|
1645
|
+
# completed target utterance into the simulator as a user turn.
|
|
1646
|
+
def on_target_transcription(
|
|
1647
|
+
reader: "rtc.TextStreamReader",
|
|
1648
|
+
participant_identity: str,
|
|
1649
|
+
) -> None:
|
|
1650
|
+
nonlocal target_transcription_mode
|
|
1651
|
+
attrs = reader.info.attributes or {}
|
|
1652
|
+
transcribed_track_id = attrs.get(ATTRIBUTE_TRANSCRIPTION_TRACK_ID)
|
|
1653
|
+
if transcribed_track_id:
|
|
1654
|
+
if transcribed_track_id != target.audio_track_sid:
|
|
1655
|
+
return
|
|
1656
|
+
elif str(participant_identity) != target.identity:
|
|
1657
|
+
return
|
|
1658
|
+
# First target transcription means the target is speaking — stop
|
|
1659
|
+
# the redundant simulator STT so it cannot emit duplicate turns.
|
|
1660
|
+
if not target_transcription_mode:
|
|
1661
|
+
session.input.set_audio_enabled(False)
|
|
1662
|
+
session.clear_user_turn()
|
|
1663
|
+
target_transcription_mode = True
|
|
1664
|
+
task = asyncio.create_task(
|
|
1665
|
+
_forward_target_transcription(
|
|
1666
|
+
reader,
|
|
1667
|
+
session,
|
|
1668
|
+
conversation_ended=conversation_ended,
|
|
1669
|
+
captured_target_turns=captured_target_turns,
|
|
1670
|
+
)
|
|
1671
|
+
)
|
|
1672
|
+
target_transcription_tasks.add(task)
|
|
1673
|
+
task.add_done_callback(target_transcription_tasks.discard)
|
|
1674
|
+
|
|
1675
|
+
room.register_text_stream_handler(
|
|
1676
|
+
TOPIC_TRANSCRIPTION, on_target_transcription
|
|
1677
|
+
)
|
|
1678
|
+
target_transcription_handler_registered = True
|
|
1679
|
+
|
|
1680
|
+
# Drain greeting streams buffered before readiness through the
|
|
1681
|
+
# authoritative handler (it re-applies the strict track/identity
|
|
1682
|
+
# filter, so over-buffered non-target streams are dropped here).
|
|
1683
|
+
buffered_streams = list(pending_target_transcriptions)
|
|
1684
|
+
pending_target_transcriptions.clear()
|
|
1685
|
+
for buffered_reader, buffered_identity in buffered_streams:
|
|
1686
|
+
on_target_transcription(buffered_reader, buffered_identity)
|
|
1687
|
+
|
|
1688
|
+
opener: asyncio.Task[None] | None = None
|
|
1689
|
+
if conversation_direction == "simulator_first" or _answered_by_voicemail():
|
|
1690
|
+
# A mailbox speaks first and needs no watchdog to break a mutual silence.
|
|
1691
|
+
customer_agent.open_conversation()
|
|
1692
|
+
else:
|
|
1693
|
+
# The agent placed this call and should speak first. If it does not, the person
|
|
1694
|
+
# answers rather than both sides waiting for each other.
|
|
1695
|
+
opener = asyncio.create_task(
|
|
1696
|
+
_open_if_nobody_speaks_first(
|
|
1697
|
+
session,
|
|
1698
|
+
customer_agent,
|
|
1699
|
+
timeout_seconds=_OPEN_INSTEAD_AFTER_SECONDS,
|
|
1700
|
+
)
|
|
1701
|
+
)
|
|
1702
|
+
try:
|
|
1703
|
+
stop_reason = await _wait_for_conversation_end(
|
|
1704
|
+
room,
|
|
1705
|
+
session,
|
|
1706
|
+
customer_agent=customer_agent,
|
|
1707
|
+
target_identity=target.identity,
|
|
1708
|
+
timeout=max_seconds,
|
|
1709
|
+
conversation_direction=conversation_direction,
|
|
1710
|
+
agent_first_silence_timeout_seconds=agent_first_silence_timeout_seconds,
|
|
1711
|
+
provider_task=bridge_task,
|
|
1712
|
+
)
|
|
1713
|
+
finally:
|
|
1714
|
+
# However the conversation ended, including badly, the watchdog goes with it: a
|
|
1715
|
+
# pending task at loop close is noise in the log of every call.
|
|
1716
|
+
if opener is not None and not opener.done():
|
|
1717
|
+
opener.cancel()
|
|
1718
|
+
logger.info(
|
|
1719
|
+
"livekit_conversation_ended stop_reason=%s run=%s case=%s",
|
|
1720
|
+
stop_reason,
|
|
1721
|
+
run_id,
|
|
1722
|
+
test_case_id,
|
|
1723
|
+
)
|
|
1724
|
+
# End the call cleanly. First let the party that just spoke commit its
|
|
1725
|
+
# own final turn — a LiveKit turn only lands in history once its TTS
|
|
1726
|
+
# finishes — bounded so we do not wait on the other side. We do NOT
|
|
1727
|
+
# wait for the target's trailing speech: once the conversation has
|
|
1728
|
+
# ended, the target talking on is monologuing into a call the other
|
|
1729
|
+
# side left.
|
|
1730
|
+
conversation_ended.set()
|
|
1731
|
+
if stop_reason == "simulator_end_call":
|
|
1732
|
+
wait_for_end_speech = getattr(
|
|
1733
|
+
customer_agent,
|
|
1734
|
+
"wait_for_end_speech",
|
|
1735
|
+
None,
|
|
1736
|
+
)
|
|
1737
|
+
try:
|
|
1738
|
+
if callable(wait_for_end_speech):
|
|
1739
|
+
await asyncio.wait_for(
|
|
1740
|
+
wait_for_end_speech(),
|
|
1741
|
+
timeout=_FINAL_TURN_COMMIT_WAIT_SECONDS,
|
|
1742
|
+
)
|
|
1743
|
+
except asyncio.TimeoutError:
|
|
1744
|
+
logger.warning(
|
|
1745
|
+
"Simulator closing speech did not finish before cleanup",
|
|
1746
|
+
extra={"run_id": run_id, "test_case_id": test_case_id},
|
|
1747
|
+
)
|
|
1748
|
+
_loop = asyncio.get_running_loop()
|
|
1749
|
+
_commit_deadline = _loop.time() + _FINAL_TURN_COMMIT_WAIT_SECONDS
|
|
1750
|
+
while _loop.time() < _commit_deadline:
|
|
1751
|
+
try:
|
|
1752
|
+
if session.current_speech is None:
|
|
1753
|
+
break
|
|
1754
|
+
except Exception: # noqa: BLE001
|
|
1755
|
+
break
|
|
1756
|
+
await asyncio.sleep(0.2)
|
|
1757
|
+
# Delete the room so the target agent can't keep monologuing into a
|
|
1758
|
+
# dead call (its audio would be recorded but is untranscribable once
|
|
1759
|
+
# the simulator has left) — the recording then ends when the call
|
|
1760
|
+
# actually ends, matching the transcript.
|
|
1761
|
+
if api_client is not None and managed_room_owned:
|
|
1762
|
+
try:
|
|
1763
|
+
await asyncio.wait_for(
|
|
1764
|
+
api_client.room.delete_room(
|
|
1765
|
+
api.DeleteRoomRequest(room=room_name)
|
|
1766
|
+
),
|
|
1767
|
+
timeout=_cleanup_budget(),
|
|
1768
|
+
)
|
|
1769
|
+
except Exception as exc: # noqa: BLE001
|
|
1770
|
+
logger.warning(
|
|
1771
|
+
"LiveKit room delete on conversation end failed",
|
|
1772
|
+
exc_info=redacted_exc_info(exc),
|
|
1773
|
+
)
|
|
1774
|
+
messages = _canonical_report_messages(session)
|
|
1775
|
+
messages = _merge_captured_target_turns(messages, captured_target_turns)
|
|
1776
|
+
outcome = _conversation_outcome(
|
|
1777
|
+
stop_reason,
|
|
1778
|
+
messages,
|
|
1779
|
+
min_turn_messages=min_turn_messages,
|
|
1780
|
+
)
|
|
1781
|
+
except asyncio.TimeoutError:
|
|
1782
|
+
stage = (
|
|
1783
|
+
FailureStage.READINESS
|
|
1784
|
+
if session is not None and target is None
|
|
1785
|
+
else FailureStage.PREPARING
|
|
1786
|
+
)
|
|
1787
|
+
if stage == FailureStage.READINESS and profile.receives_inbound_call:
|
|
1788
|
+
code = "sip_inbound_no_participant"
|
|
1789
|
+
message = "No inbound SIP participant joined before deadline"
|
|
1790
|
+
elif stage == FailureStage.READINESS:
|
|
1791
|
+
code = "agent_unavailable"
|
|
1792
|
+
message = "Target agent did not become ready"
|
|
1793
|
+
else:
|
|
1794
|
+
code = "livekit_connect_timeout"
|
|
1795
|
+
message = "LiveKit setup exceeded its deadline"
|
|
1796
|
+
status = (
|
|
1797
|
+
TestCaseStatus.AGENT_UNAVAILABLE
|
|
1798
|
+
if stage == FailureStage.READINESS
|
|
1799
|
+
else TestCaseStatus.TIMED_OUT
|
|
1800
|
+
)
|
|
1801
|
+
outcome = _failure_outcome(
|
|
1802
|
+
status,
|
|
1803
|
+
stage,
|
|
1804
|
+
code,
|
|
1805
|
+
message,
|
|
1806
|
+
retryable=True,
|
|
1807
|
+
)
|
|
1808
|
+
except _LeasedRoomCallerMismatch:
|
|
1809
|
+
# Unlike the pre-dial failures above, this path has a billed call and
|
|
1810
|
+
# a joined target; the recordings, provider evidence and metadata
|
|
1811
|
+
# update below all run after the try, so this must not return early.
|
|
1812
|
+
outcome = _failure_outcome(
|
|
1813
|
+
TestCaseStatus.FAILED,
|
|
1814
|
+
FailureStage.READINESS,
|
|
1815
|
+
"livekit_room_wrong_caller",
|
|
1816
|
+
"The participant that answered is not calling from the customer's configured number",
|
|
1817
|
+
retryable=True,
|
|
1818
|
+
)
|
|
1819
|
+
except Exception as exc:
|
|
1820
|
+
logger.error(
|
|
1821
|
+
"LiveKit test case failed",
|
|
1822
|
+
exc_info=redacted_exc_info(exc),
|
|
1823
|
+
extra={
|
|
1824
|
+
"run_id": run_id,
|
|
1825
|
+
"test_case_id": test_case_id,
|
|
1826
|
+
"exception_type": type(exc).__name__,
|
|
1827
|
+
},
|
|
1828
|
+
)
|
|
1829
|
+
outcome = _failure_outcome(
|
|
1830
|
+
TestCaseStatus.FAILED,
|
|
1831
|
+
FailureStage.RUNNING if session is not None else FailureStage.PREPARING,
|
|
1832
|
+
"livekit_case_failed",
|
|
1833
|
+
"LiveKit test case failed",
|
|
1834
|
+
details={"exception_type": type(exc).__name__},
|
|
1835
|
+
)
|
|
1836
|
+
finally:
|
|
1837
|
+
# The ambience belongs to the caller agent, not the engine. Guarded because teardown
|
|
1838
|
+
# must never be the reason a case fails.
|
|
1839
|
+
if customer_agent is not None:
|
|
1840
|
+
try:
|
|
1841
|
+
# Bounded like every other teardown step. Closing the ambience player unpublishes
|
|
1842
|
+
# its track, and when the room's signal client has already died, that wait never
|
|
1843
|
+
# returns: the SDK loops on resume and restart while this await sits here, and the
|
|
1844
|
+
# case never completes, so the whole run is discarded on its deadline with a
|
|
1845
|
+
# finished conversation inside it. Measured: two calls ended on endCall at 15 and
|
|
1846
|
+
# 17 messages and both reported 570004ms and no test case.
|
|
1847
|
+
await asyncio.wait_for(
|
|
1848
|
+
customer_agent._stop_background_audio(),
|
|
1849
|
+
timeout=_cleanup_budget(
|
|
1850
|
+
_BACKGROUND_AUDIO_CLEANUP_TIMEOUT_SECONDS
|
|
1851
|
+
),
|
|
1852
|
+
)
|
|
1853
|
+
except Exception:
|
|
1854
|
+
logger.warning("background audio not closed cleanly", exc_info=True)
|
|
1855
|
+
if target_transcription_handler_registered:
|
|
1856
|
+
room.unregister_text_stream_handler(TOPIC_TRANSCRIPTION)
|
|
1857
|
+
pending_target_transcriptions.clear()
|
|
1858
|
+
pending_transcriptions = list(target_transcription_tasks)
|
|
1859
|
+
for pending in pending_transcriptions:
|
|
1860
|
+
pending.cancel()
|
|
1861
|
+
if pending_transcriptions:
|
|
1862
|
+
# Cancelled above, but a task blocked reading a stream whose connection is gone does
|
|
1863
|
+
# not observe the cancellation, so this is bounded too.
|
|
1864
|
+
try:
|
|
1865
|
+
await asyncio.wait_for(
|
|
1866
|
+
asyncio.gather(*pending_transcriptions, return_exceptions=True),
|
|
1867
|
+
timeout=_cleanup_budget(),
|
|
1868
|
+
)
|
|
1869
|
+
except Exception as exc: # noqa: BLE001 - teardown never fails a case
|
|
1870
|
+
logger.warning("transcription tasks did not stop cleanly: %s", exc)
|
|
1871
|
+
session_to_close = session or (
|
|
1872
|
+
getattr(customer_agent, "started_session", None)
|
|
1873
|
+
if customer_agent is not None
|
|
1874
|
+
else None
|
|
1875
|
+
)
|
|
1876
|
+
if session_to_close is not None:
|
|
1877
|
+
try:
|
|
1878
|
+
await _close_agent_session(
|
|
1879
|
+
session_to_close,
|
|
1880
|
+
timeout=_cleanup_budget(_SESSION_CLEANUP_TIMEOUT_SECONDS),
|
|
1881
|
+
)
|
|
1882
|
+
except Exception as exc:
|
|
1883
|
+
_record_cleanup_error(
|
|
1884
|
+
cleanup_errors,
|
|
1885
|
+
exc,
|
|
1886
|
+
"session_close",
|
|
1887
|
+
run_id,
|
|
1888
|
+
test_case_id,
|
|
1889
|
+
)
|
|
1890
|
+
if models is not None:
|
|
1891
|
+
try:
|
|
1892
|
+
await asyncio.wait_for(
|
|
1893
|
+
models.aclose(),
|
|
1894
|
+
timeout=_cleanup_budget(),
|
|
1895
|
+
)
|
|
1896
|
+
except Exception as exc:
|
|
1897
|
+
_record_cleanup_error(
|
|
1898
|
+
cleanup_errors,
|
|
1899
|
+
exc,
|
|
1900
|
+
"models_close",
|
|
1901
|
+
run_id,
|
|
1902
|
+
test_case_id,
|
|
1903
|
+
)
|
|
1904
|
+
if recorder is not None:
|
|
1905
|
+
try:
|
|
1906
|
+
await asyncio.wait_for(
|
|
1907
|
+
recorder.aclose(),
|
|
1908
|
+
timeout=_cleanup_budget(),
|
|
1909
|
+
)
|
|
1910
|
+
except Exception as exc:
|
|
1911
|
+
_record_cleanup_error(
|
|
1912
|
+
cleanup_errors,
|
|
1913
|
+
exc,
|
|
1914
|
+
"recorder_close",
|
|
1915
|
+
run_id,
|
|
1916
|
+
test_case_id,
|
|
1917
|
+
)
|
|
1918
|
+
if audio_bridge is not None:
|
|
1919
|
+
try:
|
|
1920
|
+
await asyncio.wait_for(
|
|
1921
|
+
audio_bridge.aclose(), timeout=_cleanup_budget()
|
|
1922
|
+
)
|
|
1923
|
+
if bridge_task is not None:
|
|
1924
|
+
await asyncio.wait_for(bridge_task, timeout=_cleanup_budget())
|
|
1925
|
+
except Exception as exc:
|
|
1926
|
+
_record_cleanup_error(
|
|
1927
|
+
cleanup_errors,
|
|
1928
|
+
exc,
|
|
1929
|
+
"bridge_close",
|
|
1930
|
+
run_id,
|
|
1931
|
+
test_case_id,
|
|
1932
|
+
)
|
|
1933
|
+
if room_connected:
|
|
1934
|
+
try:
|
|
1935
|
+
await asyncio.wait_for(room.disconnect(), timeout=_cleanup_budget())
|
|
1936
|
+
except Exception as exc:
|
|
1937
|
+
_record_cleanup_error(
|
|
1938
|
+
cleanup_errors,
|
|
1939
|
+
exc,
|
|
1940
|
+
"room_disconnect",
|
|
1941
|
+
run_id,
|
|
1942
|
+
test_case_id,
|
|
1943
|
+
)
|
|
1944
|
+
if call_originator is not None:
|
|
1945
|
+
# Guarded like every other cleanup step below: a future escape
|
|
1946
|
+
# from the helper (today it never raises) must not skip the
|
|
1947
|
+
# dispatch-rule/room/api-client teardown that follows.
|
|
1948
|
+
try:
|
|
1949
|
+
finalize_result = await finalize_originator(
|
|
1950
|
+
call_originator,
|
|
1951
|
+
provider_call_id=provider_call_id,
|
|
1952
|
+
originator_name=transport.inbound_call_originator,
|
|
1953
|
+
case_started_at=case_started_at,
|
|
1954
|
+
cleanup_timeout=_cleanup_budget(),
|
|
1955
|
+
)
|
|
1956
|
+
for operation, exc in finalize_result.cleanup_errors:
|
|
1957
|
+
_record_cleanup_error(
|
|
1958
|
+
cleanup_errors, exc, operation, run_id, test_case_id
|
|
1959
|
+
)
|
|
1960
|
+
if finalize_result.termination_source:
|
|
1961
|
+
provider_termination_source = finalize_result.termination_source
|
|
1962
|
+
reconciled_call_ids = finalize_result.reconciled_call_ids
|
|
1963
|
+
if provider_call_id is None and reconciled_call_ids:
|
|
1964
|
+
# We stopped a billed call we believe is ours, so
|
|
1965
|
+
# evidence fetches its record instead of reporting
|
|
1966
|
+
# "not matched" after searching nothing.
|
|
1967
|
+
provider_call_id = reconciled_call_ids[0]
|
|
1968
|
+
# Written here too (not only in the metadata.update below)
|
|
1969
|
+
# because a start-failure path returns from inside the try,
|
|
1970
|
+
# bypassing that block entirely.
|
|
1971
|
+
if outcome is not None:
|
|
1972
|
+
outcome.metadata["reconciled_call_ids"] = reconciled_call_ids
|
|
1973
|
+
if finalize_result.termination_source:
|
|
1974
|
+
outcome.metadata["provider_termination_source"] = (
|
|
1975
|
+
finalize_result.termination_source
|
|
1976
|
+
)
|
|
1977
|
+
except Exception as exc:
|
|
1978
|
+
_record_cleanup_error(
|
|
1979
|
+
cleanup_errors,
|
|
1980
|
+
exc,
|
|
1981
|
+
f"{transport.inbound_call_originator}_call_finalize",
|
|
1982
|
+
run_id,
|
|
1983
|
+
test_case_id,
|
|
1984
|
+
)
|
|
1985
|
+
if (
|
|
1986
|
+
api_client is not None
|
|
1987
|
+
and sip_dispatch_rule_id
|
|
1988
|
+
and sip_dispatch_rule_created
|
|
1989
|
+
):
|
|
1990
|
+
try:
|
|
1991
|
+
await asyncio.wait_for(
|
|
1992
|
+
_delete_sip_dispatch_rule(api_client, sip_dispatch_rule_id),
|
|
1993
|
+
timeout=_cleanup_budget(),
|
|
1994
|
+
)
|
|
1995
|
+
except Exception as exc:
|
|
1996
|
+
if not _is_not_found(exc):
|
|
1997
|
+
_record_cleanup_error(
|
|
1998
|
+
cleanup_errors,
|
|
1999
|
+
exc,
|
|
2000
|
+
"sip_dispatch_delete",
|
|
2001
|
+
run_id,
|
|
2002
|
+
test_case_id,
|
|
2003
|
+
)
|
|
2004
|
+
if api_client is not None and managed_room_owned:
|
|
2005
|
+
try:
|
|
2006
|
+
await asyncio.wait_for(
|
|
2007
|
+
api_client.room.delete_room(
|
|
2008
|
+
api.DeleteRoomRequest(room=room_name)
|
|
2009
|
+
),
|
|
2010
|
+
timeout=_cleanup_budget(),
|
|
2011
|
+
)
|
|
2012
|
+
except Exception as exc:
|
|
2013
|
+
if not _is_not_found(exc):
|
|
2014
|
+
_record_cleanup_error(
|
|
2015
|
+
cleanup_errors,
|
|
2016
|
+
exc,
|
|
2017
|
+
"room_delete",
|
|
2018
|
+
run_id,
|
|
2019
|
+
test_case_id,
|
|
2020
|
+
)
|
|
2021
|
+
if api_client is not None:
|
|
2022
|
+
try:
|
|
2023
|
+
await api_client.aclose()
|
|
2024
|
+
except Exception as exc:
|
|
2025
|
+
_record_cleanup_error(
|
|
2026
|
+
cleanup_errors,
|
|
2027
|
+
exc,
|
|
2028
|
+
"api_close",
|
|
2029
|
+
run_id,
|
|
2030
|
+
test_case_id,
|
|
2031
|
+
)
|
|
2032
|
+
# Written here too (not only in the metadata.update below) because a
|
|
2033
|
+
# pre-dial return (e.g. the occupancy check) exits from inside the
|
|
2034
|
+
# try, bypassing that block entirely.
|
|
2035
|
+
if outcome is not None and outcome.metadata.get("cleanup_status") is None:
|
|
2036
|
+
outcome.metadata["cleanup_status"] = (
|
|
2037
|
+
"failed" if cleanup_errors else "completed"
|
|
2038
|
+
)
|
|
2039
|
+
outcome.metadata["cleanup_errors"] = cleanup_errors
|
|
2040
|
+
if outcome is None:
|
|
2041
|
+
outcome = _failure_outcome(
|
|
2042
|
+
TestCaseStatus.FAILED,
|
|
2043
|
+
FailureStage.FINALIZING,
|
|
2044
|
+
"livekit_outcome_missing",
|
|
2045
|
+
"LiveKit test case ended without an outcome",
|
|
2046
|
+
)
|
|
2047
|
+
if recorder is not None:
|
|
2048
|
+
_attach_recordings(
|
|
2049
|
+
outcome,
|
|
2050
|
+
recorder,
|
|
2051
|
+
simulator_identity=simulator_identity,
|
|
2052
|
+
target_identity=target.identity if target is not None else None,
|
|
2053
|
+
target_track_sid=(
|
|
2054
|
+
target.audio_track_sid if target is not None else None
|
|
2055
|
+
),
|
|
2056
|
+
case_directory=case_directory,
|
|
2057
|
+
sample_rate=recorder_sample_rate,
|
|
2058
|
+
)
|
|
2059
|
+
if recorder.errors:
|
|
2060
|
+
cleanup_errors.extend(
|
|
2061
|
+
f"recording:{type(error).__name__}" for error in recorder.errors
|
|
2062
|
+
)
|
|
2063
|
+
if agent_definition.provider_evidence is not None:
|
|
2064
|
+
provider_summary, provider_artifacts = await _collect_provider_evidence(
|
|
2065
|
+
config=agent_definition.provider_evidence,
|
|
2066
|
+
transport=transport,
|
|
2067
|
+
run_id=run_id,
|
|
2068
|
+
test_case_id=test_case_id,
|
|
2069
|
+
case_directory=case_directory,
|
|
2070
|
+
started_at=case_started_at,
|
|
2071
|
+
target=target,
|
|
2072
|
+
provider_call_id_hint=provider_call_id,
|
|
2073
|
+
provider_api_key=_target_api_key(provider_target),
|
|
2074
|
+
provider_api_base_url=_target_evidence_base_url(provider_target),
|
|
2075
|
+
termination_source=provider_termination_source,
|
|
2076
|
+
)
|
|
2077
|
+
if provider_summary is not None:
|
|
2078
|
+
outcome.evidence.append(provider_summary)
|
|
2079
|
+
if provider_call_id is None:
|
|
2080
|
+
resolved_call_id = provider_summary.metadata.get("call_id")
|
|
2081
|
+
if resolved_call_id:
|
|
2082
|
+
provider_call_id = str(resolved_call_id)
|
|
2083
|
+
_reconcile_provider_observation(outcome, provider_summary)
|
|
2084
|
+
_recover_successful_provider_end_call(outcome, provider_summary)
|
|
2085
|
+
outcome.provider_artifacts.extend(provider_artifacts)
|
|
2086
|
+
outcome.metadata.update(
|
|
2087
|
+
{
|
|
2088
|
+
"simulator_participant_identity": simulator_identity,
|
|
2089
|
+
"target_participant_identity": (
|
|
2090
|
+
target.identity if target is not None else None
|
|
2091
|
+
),
|
|
2092
|
+
"target_participant_sid": target.sid if target is not None else None,
|
|
2093
|
+
"target_audio_track_sid": (
|
|
2094
|
+
target.audio_track_sid if target is not None else None
|
|
2095
|
+
),
|
|
2096
|
+
"target_participant_attributes": (
|
|
2097
|
+
dict(target.attributes) if target is not None else {}
|
|
2098
|
+
),
|
|
2099
|
+
"cleanup_status": "failed" if cleanup_errors else "completed",
|
|
2100
|
+
"cleanup_errors": cleanup_errors,
|
|
2101
|
+
"sip_dispatch_rule_id": sip_dispatch_rule_id,
|
|
2102
|
+
"sip_dispatch_rule_created": sip_dispatch_rule_created,
|
|
2103
|
+
"target_provider": (
|
|
2104
|
+
provider_target.provider if provider_target is not None else None
|
|
2105
|
+
),
|
|
2106
|
+
"provider_call_id": provider_call_id,
|
|
2107
|
+
"reconciled_call_ids": reconciled_call_ids,
|
|
2108
|
+
"caller_verification": caller_verification,
|
|
2109
|
+
"vapi_call_id": (
|
|
2110
|
+
provider_call_id
|
|
2111
|
+
if profile.evidence_provider == "vapi"
|
|
2112
|
+
or transport.inbound_call_originator == "vapi"
|
|
2113
|
+
else None
|
|
2114
|
+
),
|
|
2115
|
+
"retell_call_id": (
|
|
2116
|
+
provider_call_id
|
|
2117
|
+
if profile.evidence_provider == "retell"
|
|
2118
|
+
or transport.inbound_call_originator == "retell"
|
|
2119
|
+
else None
|
|
2120
|
+
),
|
|
2121
|
+
"simulator_model_usage": (
|
|
2122
|
+
customer_agent.model_usage
|
|
2123
|
+
if customer_agent is not None
|
|
2124
|
+
and hasattr(customer_agent, "model_usage")
|
|
2125
|
+
else []
|
|
2126
|
+
),
|
|
2127
|
+
}
|
|
2128
|
+
)
|
|
2129
|
+
logger.info(
|
|
2130
|
+
"livekit_case_outcome status=%s stop_reason=%s failure=%s run=%s case=%s",
|
|
2131
|
+
outcome.status.value,
|
|
2132
|
+
outcome.metadata.get("stop_reason"),
|
|
2133
|
+
outcome.failure.code if outcome.failure is not None else None,
|
|
2134
|
+
run_id,
|
|
2135
|
+
test_case_id,
|
|
2136
|
+
)
|
|
2137
|
+
return outcome
|
|
2138
|
+
|
|
2139
|
+
async def _create_customer_agent(
|
|
2140
|
+
self,
|
|
2141
|
+
persona: Persona,
|
|
2142
|
+
simulator: SimulatorAgentDefinition | None,
|
|
2143
|
+
*,
|
|
2144
|
+
call_type: CallType = "inbound",
|
|
2145
|
+
agent_name: str | None = None,
|
|
2146
|
+
min_turn_messages: int = 0,
|
|
2147
|
+
) -> tuple[_TestRunnerAgent, LiveKitModels]:
|
|
2148
|
+
customer_prompt = build_voice_simulator_prompt(
|
|
2149
|
+
persona,
|
|
2150
|
+
call_type=call_type,
|
|
2151
|
+
agent_name=agent_name,
|
|
2152
|
+
additional_instructions=(
|
|
2153
|
+
simulator.instructions if simulator is not None else None
|
|
2154
|
+
),
|
|
2155
|
+
default_language=(
|
|
2156
|
+
simulator.stt.language if simulator is not None else None
|
|
2157
|
+
),
|
|
2158
|
+
variables={"instruction": persona.situation or ""},
|
|
2159
|
+
# Delivery cues are Cartesia only. Passing the provider here rather than reading it
|
|
2160
|
+
# inside the prompt keeps the decision where the provider is actually known.
|
|
2161
|
+
tts_provider=(simulator.tts.provider if simulator is not None else None),
|
|
2162
|
+
)
|
|
2163
|
+
if simulator is None:
|
|
2164
|
+
voice_provider = os.environ.get(
|
|
2165
|
+
"SIMULATOR_VOICE_PROVIDER", "openai"
|
|
2166
|
+
).lower()
|
|
2167
|
+
llm_config = _default_simulator_llm_config()
|
|
2168
|
+
stt_config = STTConfig(
|
|
2169
|
+
provider=voice_provider,
|
|
2170
|
+
model=os.environ.get("SIMULATOR_STT_MODEL", "gpt-4o-mini-transcribe"),
|
|
2171
|
+
)
|
|
2172
|
+
tts_config = TTSConfig(
|
|
2173
|
+
provider=voice_provider,
|
|
2174
|
+
model=os.environ.get("SIMULATOR_TTS_MODEL", "gpt-4o-mini-tts"),
|
|
2175
|
+
voice=os.environ.get("SIMULATOR_TTS_VOICE_ID", "alloy"),
|
|
2176
|
+
)
|
|
2177
|
+
instructions = customer_prompt
|
|
2178
|
+
allow_interruptions = None
|
|
2179
|
+
min_endpointing_delay = None
|
|
2180
|
+
max_endpointing_delay = None
|
|
2181
|
+
use_aligned_transcript = None
|
|
2182
|
+
else:
|
|
2183
|
+
llm_config = simulator.llm
|
|
2184
|
+
stt_config = simulator.stt
|
|
2185
|
+
tts_config = simulator.tts
|
|
2186
|
+
instructions = customer_prompt
|
|
2187
|
+
allow_interruptions = simulator.allow_interruptions
|
|
2188
|
+
min_endpointing_delay = simulator.min_endpointing_delay
|
|
2189
|
+
max_endpointing_delay = simulator.max_endpointing_delay
|
|
2190
|
+
use_aligned_transcript = simulator.use_tts_aligned_transcript
|
|
2191
|
+
# Per-persona voice: a persona may carry a ``voice`` (or ``voice_id``)
|
|
2192
|
+
# attribute so different simulated customers sound different. Deepgram
|
|
2193
|
+
# encodes the voice in the model name (``aura-*``); every other provider
|
|
2194
|
+
# uses the dedicated ``voice`` field. Falls back to the simulator/global
|
|
2195
|
+
# default when the persona does not specify one.
|
|
2196
|
+
persona_attrs = getattr(persona, "persona", None)
|
|
2197
|
+
persona_voice = (
|
|
2198
|
+
(persona_attrs.get("voice") or persona_attrs.get("voice_id"))
|
|
2199
|
+
if isinstance(persona_attrs, dict)
|
|
2200
|
+
else None
|
|
2201
|
+
)
|
|
2202
|
+
if persona_voice:
|
|
2203
|
+
field = "model" if tts_config.provider == "deepgram" else "voice"
|
|
2204
|
+
tts_config = tts_config.model_copy(update={field: str(persona_voice)})
|
|
2205
|
+
models = await build_livekit_models(
|
|
2206
|
+
llm_config=llm_config,
|
|
2207
|
+
stt_config=stt_config,
|
|
2208
|
+
tts_config=tts_config,
|
|
2209
|
+
)
|
|
2210
|
+
vad = await asyncio.to_thread(_load_silero_vad_sync)
|
|
2211
|
+
self._last_simulator_setup = {
|
|
2212
|
+
"instructions": instructions,
|
|
2213
|
+
"llm_config": llm_config,
|
|
2214
|
+
"stt_config": stt_config,
|
|
2215
|
+
"tts_config": tts_config,
|
|
2216
|
+
"allow_interruptions": allow_interruptions,
|
|
2217
|
+
"min_endpointing_delay": min_endpointing_delay,
|
|
2218
|
+
"max_endpointing_delay": max_endpointing_delay,
|
|
2219
|
+
"use_tts_aligned_transcript": use_aligned_transcript,
|
|
2220
|
+
}
|
|
2221
|
+
agent = _TestRunnerAgent(
|
|
2222
|
+
persona=persona,
|
|
2223
|
+
min_turn_messages=min_turn_messages,
|
|
2224
|
+
stt=models.stt,
|
|
2225
|
+
llm=models.llm,
|
|
2226
|
+
tts=models.tts,
|
|
2227
|
+
vad=vad,
|
|
2228
|
+
instructions=instructions,
|
|
2229
|
+
turn_handling=_simulator_turn_handling(
|
|
2230
|
+
vad=vad,
|
|
2231
|
+
allow_interruptions=allow_interruptions,
|
|
2232
|
+
min_endpointing_delay=min_endpointing_delay,
|
|
2233
|
+
max_endpointing_delay=max_endpointing_delay,
|
|
2234
|
+
),
|
|
2235
|
+
use_tts_aligned_transcript=use_aligned_transcript,
|
|
2236
|
+
)
|
|
2237
|
+
return agent, models
|
|
2238
|
+
|
|
2239
|
+
|
|
2240
|
+
async def _wait_for_target_audio(
|
|
2241
|
+
room: rtc.Room,
|
|
2242
|
+
*,
|
|
2243
|
+
excluded_identities: set[str],
|
|
2244
|
+
target_identity: str | None,
|
|
2245
|
+
timeout: float,
|
|
2246
|
+
) -> _TargetParticipant:
|
|
2247
|
+
ready = asyncio.Event()
|
|
2248
|
+
selected: _TargetParticipant | None = None
|
|
2249
|
+
|
|
2250
|
+
def inspect_room(*_args) -> None:
|
|
2251
|
+
nonlocal selected
|
|
2252
|
+
selected = _find_target_audio(
|
|
2253
|
+
room,
|
|
2254
|
+
excluded_identities=excluded_identities,
|
|
2255
|
+
target_identity=target_identity,
|
|
2256
|
+
)
|
|
2257
|
+
if selected is not None:
|
|
2258
|
+
ready.set()
|
|
2259
|
+
|
|
2260
|
+
room.on("participant_connected", inspect_room)
|
|
2261
|
+
room.on("track_published", inspect_room)
|
|
2262
|
+
room.on("track_subscribed", inspect_room)
|
|
2263
|
+
inspect_room()
|
|
2264
|
+
try:
|
|
2265
|
+
await asyncio.wait_for(ready.wait(), timeout=timeout)
|
|
2266
|
+
finally:
|
|
2267
|
+
_remove_room_listener(room, "participant_connected", inspect_room)
|
|
2268
|
+
_remove_room_listener(room, "track_published", inspect_room)
|
|
2269
|
+
_remove_room_listener(room, "track_subscribed", inspect_room)
|
|
2270
|
+
if selected is None:
|
|
2271
|
+
raise asyncio.TimeoutError
|
|
2272
|
+
return selected
|
|
2273
|
+
|
|
2274
|
+
|
|
2275
|
+
async def _forward_target_transcription(
|
|
2276
|
+
reader: "rtc.TextStreamReader",
|
|
2277
|
+
session: "AgentSession",
|
|
2278
|
+
*,
|
|
2279
|
+
conversation_ended: "asyncio.Event | None" = None,
|
|
2280
|
+
captured_target_turns: list[dict[str, Any]] | None = None,
|
|
2281
|
+
) -> None:
|
|
2282
|
+
# Receiver-side wall clock — same clock domain as the simulator's
|
|
2283
|
+
# ChatMessage.metrics, and the target's transcript IO is playback-synced
|
|
2284
|
+
# (TranscriptSynchronizer), so stream-open ~= speech start and read_all()
|
|
2285
|
+
# completion ~= speech end. Timestamps embedded in the stream are the
|
|
2286
|
+
# sender's (laptop) clock; skew there would corrupt the derived latencies.
|
|
2287
|
+
started_at = time.time()
|
|
2288
|
+
try:
|
|
2289
|
+
transcript = (await reader.read_all()).strip()
|
|
2290
|
+
stopped_at = time.time()
|
|
2291
|
+
if not transcript:
|
|
2292
|
+
return
|
|
2293
|
+
# Capture the target's turn independently of the simulator session FIRST.
|
|
2294
|
+
# Once the session drains it rejects new input ("speech scheduling is
|
|
2295
|
+
# paused"), so a closing delivered after the simulator is done never
|
|
2296
|
+
# reaches the chat context. This list is merged into the report so the
|
|
2297
|
+
# trailing target turn survives regardless of session state.
|
|
2298
|
+
if captured_target_turns is not None:
|
|
2299
|
+
captured_target_turns.append(
|
|
2300
|
+
{
|
|
2301
|
+
"content": transcript,
|
|
2302
|
+
"started_speaking_at": started_at,
|
|
2303
|
+
"stopped_speaking_at": stopped_at,
|
|
2304
|
+
}
|
|
2305
|
+
)
|
|
2306
|
+
# Only elicit a simulator response while the conversation is live; once
|
|
2307
|
+
# it has ended the target's turn is recorded but the simulator stays
|
|
2308
|
+
# silent. The turn MUST travel through ``generate_reply(user_input=...)``:
|
|
2309
|
+
# the reply pipeline reads the agent's own chat context, not
|
|
2310
|
+
# ``session.history``, so a turn only added to the history is invisible
|
|
2311
|
+
# to the simulator LLM (it answers as if it heard nothing). The pipeline
|
|
2312
|
+
# then persists the message into both contexts once the reply schedules.
|
|
2313
|
+
if conversation_ended is None or not conversation_ended.is_set():
|
|
2314
|
+
try:
|
|
2315
|
+
session.generate_reply(user_input=transcript)
|
|
2316
|
+
except RuntimeError:
|
|
2317
|
+
# Session is already closing; the turn is captured above.
|
|
2318
|
+
pass
|
|
2319
|
+
else:
|
|
2320
|
+
return
|
|
2321
|
+
# Conversation over (or the session rejected the reply): record the
|
|
2322
|
+
# turn on the transcript without eliciting a response.
|
|
2323
|
+
try:
|
|
2324
|
+
session.history.add_message(role="user", content=transcript)
|
|
2325
|
+
except Exception: # noqa: BLE001
|
|
2326
|
+
pass
|
|
2327
|
+
except Exception as exc: # noqa: BLE001
|
|
2328
|
+
logger.warning(
|
|
2329
|
+
"Failed to consume target transcription stream",
|
|
2330
|
+
exc_info=redacted_exc_info(exc),
|
|
2331
|
+
)
|
|
2332
|
+
|
|
2333
|
+
|
|
2334
|
+
def _find_target_audio(
|
|
2335
|
+
room: rtc.Room,
|
|
2336
|
+
*,
|
|
2337
|
+
excluded_identities: set[str],
|
|
2338
|
+
target_identity: str | None,
|
|
2339
|
+
) -> _TargetParticipant | None:
|
|
2340
|
+
candidates: list[tuple[int, int, _TargetParticipant]] = []
|
|
2341
|
+
agent_kind = getattr(
|
|
2342
|
+
rtc.ParticipantKind,
|
|
2343
|
+
"PARTICIPANT_KIND_AGENT",
|
|
2344
|
+
None,
|
|
2345
|
+
)
|
|
2346
|
+
for participant in room.remote_participants.values():
|
|
2347
|
+
identity = str(participant.identity)
|
|
2348
|
+
if identity in excluded_identities:
|
|
2349
|
+
continue
|
|
2350
|
+
if target_identity is not None and identity != target_identity:
|
|
2351
|
+
continue
|
|
2352
|
+
priority = 0 if getattr(participant, "kind", None) == agent_kind else 1
|
|
2353
|
+
for publication in participant.track_publications.values():
|
|
2354
|
+
if getattr(publication, "kind", None) != rtc.TrackKind.KIND_AUDIO:
|
|
2355
|
+
continue
|
|
2356
|
+
# Agents may publish ambient music/noise alongside their synthesized speech. The
|
|
2357
|
+
# first LiveKit publication is not necessarily the conversational track (the
|
|
2358
|
+
# official drive-thru example publishes ``background_audio``). Prefer ordinary
|
|
2359
|
+
# speech/microphone tracks so STT, transcription filtering, and recording all bind
|
|
2360
|
+
# to the same semantic stream.
|
|
2361
|
+
track_name = str(getattr(publication, "name", "") or "").lower()
|
|
2362
|
+
background = any(
|
|
2363
|
+
marker in track_name
|
|
2364
|
+
for marker in ("background", "ambient", "music", "sound_effect")
|
|
2365
|
+
)
|
|
2366
|
+
track_priority = 1 if background else 0
|
|
2367
|
+
attrs = dict(getattr(participant, "attributes", {}) or {})
|
|
2368
|
+
candidates.append(
|
|
2369
|
+
(
|
|
2370
|
+
priority,
|
|
2371
|
+
track_priority,
|
|
2372
|
+
_TargetParticipant(
|
|
2373
|
+
identity=identity,
|
|
2374
|
+
sid=str(participant.sid),
|
|
2375
|
+
audio_track_sid=str(publication.sid),
|
|
2376
|
+
attributes={
|
|
2377
|
+
str(key): str(value) for key, value in attrs.items()
|
|
2378
|
+
},
|
|
2379
|
+
),
|
|
2380
|
+
)
|
|
2381
|
+
)
|
|
2382
|
+
if not candidates:
|
|
2383
|
+
return None
|
|
2384
|
+
return sorted(
|
|
2385
|
+
candidates,
|
|
2386
|
+
key=lambda item: (
|
|
2387
|
+
item[0],
|
|
2388
|
+
item[1],
|
|
2389
|
+
item[2].identity,
|
|
2390
|
+
item[2].audio_track_sid,
|
|
2391
|
+
),
|
|
2392
|
+
)[0][2]
|
|
2393
|
+
|
|
2394
|
+
|
|
2395
|
+
# A run ends naturally when the simulator calls ``endCall``; this is only the
|
|
2396
|
+
# backstop for a conversation that has genuinely stalled or already finished but
|
|
2397
|
+
# never hung up. Kept long so normal turn-gaps (STT endpoint + LLM + TTS latency)
|
|
2398
|
+
# never trip it — the run is never cut off at a message count.
|
|
2399
|
+
_SILENCE_BACKSTOP_SECONDS = 90.0
|
|
2400
|
+
|
|
2401
|
+
# Mutual silence in a conversation both sides joined is a finished call, not a stalled one. A
|
|
2402
|
+
# thinking agent is working rather than silent, so the timer holds while either side is busy: no
|
|
2403
|
+
# fixed window fits both a 4.3s and a 25.2s reply. The measured fallback covers providers that
|
|
2404
|
+
# report no thinking state, stretching to the slowest reply this call has seen.
|
|
2405
|
+
_SETTLED_SILENCE_FLOOR_SECONDS = 12.0
|
|
2406
|
+
_SETTLED_LATENCY_MULTIPLE = 2.0
|
|
2407
|
+
|
|
2408
|
+
# LiveKit reports OUR SIMULATED CALLER as "assistant" and the TARGET AGENT as "user", because the
|
|
2409
|
+
# caller is this session's agent and the target connects as the remote party. The published
|
|
2410
|
+
# transcript swaps them (see _canonical_report_messages), so session-native code must never reuse
|
|
2411
|
+
# the published convention. Named here because reading it the wrong way round is silent: a check
|
|
2412
|
+
# still runs, still passes its tests, and watches the wrong side of the call.
|
|
2413
|
+
_CALLER = "assistant"
|
|
2414
|
+
_TARGET = "user"
|
|
2415
|
+
|
|
2416
|
+
# AgentState describes THIS SESSION'S AGENT, our caller; UserState describes the target. UserState
|
|
2417
|
+
# has no "thinking", so this pair stops us cutting off our own caller mid-thought and cannot see a
|
|
2418
|
+
# target composing a reply. The measured window below is what protects a slow target.
|
|
2419
|
+
_AGENT_BUSY_STATES = frozenset({"initializing", "thinking", "speaking"})
|
|
2420
|
+
_USER_BUSY_STATES = frozenset({"speaking"})
|
|
2421
|
+
|
|
2422
|
+
|
|
2423
|
+
def _either_side_busy(session: Any) -> bool:
|
|
2424
|
+
"""Whether work is in flight, as opposed to a conversation that has gone quiet."""
|
|
2425
|
+
return (
|
|
2426
|
+
getattr(session, "agent_state", None) in _AGENT_BUSY_STATES
|
|
2427
|
+
or getattr(session, "user_state", None) in _USER_BUSY_STATES
|
|
2428
|
+
)
|
|
2429
|
+
|
|
2430
|
+
|
|
2431
|
+
def _settled_silence_window(
|
|
2432
|
+
observed_agent_reply: float, backstop_seconds: float
|
|
2433
|
+
) -> float:
|
|
2434
|
+
"""How long silence must last before a settled call is treated as over."""
|
|
2435
|
+
return min(
|
|
2436
|
+
backstop_seconds,
|
|
2437
|
+
max(
|
|
2438
|
+
_SETTLED_SILENCE_FLOOR_SECONDS,
|
|
2439
|
+
observed_agent_reply * _SETTLED_LATENCY_MULTIPLE,
|
|
2440
|
+
),
|
|
2441
|
+
)
|
|
2442
|
+
|
|
2443
|
+
|
|
2444
|
+
def _turn_gap_seconds(
|
|
2445
|
+
previous: dict[str, Any], current: dict[str, Any]
|
|
2446
|
+
) -> float | None:
|
|
2447
|
+
"""Silence between one turn finishing and the next starting, in seconds.
|
|
2448
|
+
|
|
2449
|
+
Uses the real audio timing the transport reports and falls back to the wall-clock stamp for
|
|
2450
|
+
text-only turns. Returns None when neither side is timed, so a caller can tell "no gap" from
|
|
2451
|
+
"not measurable" rather than reading an absent measurement as zero.
|
|
2452
|
+
"""
|
|
2453
|
+
start = current.get("started_speaking_at") or current.get("created_at") or None
|
|
2454
|
+
end = previous.get("stopped_speaking_at") or previous.get("created_at") or None
|
|
2455
|
+
if not start or not end:
|
|
2456
|
+
return None
|
|
2457
|
+
gap = float(start) - float(end)
|
|
2458
|
+
return gap if gap >= 0 else None
|
|
2459
|
+
|
|
2460
|
+
|
|
2461
|
+
def _observed_agent_reply_seconds(messages: list[dict[str, Any]]) -> float:
|
|
2462
|
+
"""The slowest reply this agent has actually produced on this call.
|
|
2463
|
+
|
|
2464
|
+
Read from the transport rather than tracked against the poll loop's own clock, so it is also
|
|
2465
|
+
correct for history that arrives in bulk (a resume, a reconnect) where there was no live
|
|
2466
|
+
transition to observe. Prefers LiveKit's reported end-to-end latency and falls back to the gap
|
|
2467
|
+
between the caller finishing and the agent starting, which is the wait a listener would hear.
|
|
2468
|
+
"""
|
|
2469
|
+
slowest = 0.0
|
|
2470
|
+
previous: dict[str, Any] | None = None
|
|
2471
|
+
for message in messages:
|
|
2472
|
+
if not message.get("content"):
|
|
2473
|
+
continue
|
|
2474
|
+
if message.get("role") == _TARGET:
|
|
2475
|
+
reported = message.get("e2e_latency")
|
|
2476
|
+
if reported:
|
|
2477
|
+
slowest = max(slowest, float(reported))
|
|
2478
|
+
elif previous is not None and previous.get("role") == _CALLER:
|
|
2479
|
+
gap = _turn_gap_seconds(previous, message)
|
|
2480
|
+
if gap is not None:
|
|
2481
|
+
slowest = max(slowest, gap)
|
|
2482
|
+
previous = message
|
|
2483
|
+
return slowest
|
|
2484
|
+
|
|
2485
|
+
|
|
2486
|
+
async def _wait_for_conversation_end(
|
|
2487
|
+
room: rtc.Room,
|
|
2488
|
+
session: AgentSession,
|
|
2489
|
+
*,
|
|
2490
|
+
customer_agent: _TestRunnerAgent,
|
|
2491
|
+
target_identity: str,
|
|
2492
|
+
timeout: float,
|
|
2493
|
+
conversation_direction: str,
|
|
2494
|
+
agent_first_silence_timeout_seconds: float,
|
|
2495
|
+
provider_task: asyncio.Task[None] | None = None,
|
|
2496
|
+
) -> str:
|
|
2497
|
+
closed = asyncio.Event()
|
|
2498
|
+
target_disconnected = asyncio.Event()
|
|
2499
|
+
room_disconnected = asyncio.Event()
|
|
2500
|
+
|
|
2501
|
+
def on_close(_event) -> None:
|
|
2502
|
+
closed.set()
|
|
2503
|
+
|
|
2504
|
+
def on_participant_disconnected(participant) -> None:
|
|
2505
|
+
if str(participant.identity) == target_identity:
|
|
2506
|
+
target_disconnected.set()
|
|
2507
|
+
|
|
2508
|
+
def on_room_disconnected(*_args) -> None:
|
|
2509
|
+
room_disconnected.set()
|
|
2510
|
+
|
|
2511
|
+
session.on("close", on_close)
|
|
2512
|
+
room.on("participant_disconnected", on_participant_disconnected)
|
|
2513
|
+
# A native target commonly hangs up by DELETING the room (the LiveKit
|
|
2514
|
+
# hangup recipe); the simulator then sees a room disconnect, not a
|
|
2515
|
+
# participant_disconnected, and without this watcher the case idled
|
|
2516
|
+
# through the silence backstop before ending.
|
|
2517
|
+
room.on("disconnected", on_room_disconnected)
|
|
2518
|
+
# The target may have left in the gap between readiness and this
|
|
2519
|
+
# registration — the event is gone, so recheck presence once.
|
|
2520
|
+
remote_participants = getattr(room, "remote_participants", None)
|
|
2521
|
+
if isinstance(remote_participants, dict) and not any(
|
|
2522
|
+
str(participant.identity) == target_identity
|
|
2523
|
+
for participant in remote_participants.values()
|
|
2524
|
+
):
|
|
2525
|
+
target_disconnected.set()
|
|
2526
|
+
tasks = {
|
|
2527
|
+
"closed": asyncio.create_task(closed.wait()),
|
|
2528
|
+
"target_disconnected": asyncio.create_task(target_disconnected.wait()),
|
|
2529
|
+
"room_disconnected": asyncio.create_task(room_disconnected.wait()),
|
|
2530
|
+
"simulator_end_call": asyncio.create_task(customer_agent.end_requested.wait()),
|
|
2531
|
+
"conversation_stalled": asyncio.create_task(
|
|
2532
|
+
_wait_for_conversation_silence(
|
|
2533
|
+
session,
|
|
2534
|
+
# A stub agent in a test carries no floor; absent means never settle early.
|
|
2535
|
+
min_turn_messages=int(
|
|
2536
|
+
getattr(customer_agent, "_min_turn_messages", 0) or 0
|
|
2537
|
+
),
|
|
2538
|
+
)
|
|
2539
|
+
),
|
|
2540
|
+
"closing_loop": asyncio.create_task(_wait_for_closing_loop(session)),
|
|
2541
|
+
"no_conversation": asyncio.create_task(
|
|
2542
|
+
_wait_for_conversation_never_started(
|
|
2543
|
+
session,
|
|
2544
|
+
timeout_seconds=_NO_CONVERSATION_TIMEOUT_SECONDS,
|
|
2545
|
+
)
|
|
2546
|
+
),
|
|
2547
|
+
}
|
|
2548
|
+
if conversation_direction == "agent_first":
|
|
2549
|
+
tasks["conversation_silence_timeout"] = asyncio.create_task(
|
|
2550
|
+
_wait_for_agent_first_silence(
|
|
2551
|
+
session,
|
|
2552
|
+
timeout_seconds=agent_first_silence_timeout_seconds,
|
|
2553
|
+
)
|
|
2554
|
+
)
|
|
2555
|
+
if provider_task is not None:
|
|
2556
|
+
tasks["provider_disconnected"] = provider_task
|
|
2557
|
+
try:
|
|
2558
|
+
done, pending = await asyncio.wait(
|
|
2559
|
+
set(tasks.values()),
|
|
2560
|
+
timeout=timeout,
|
|
2561
|
+
return_when=asyncio.FIRST_COMPLETED,
|
|
2562
|
+
)
|
|
2563
|
+
owned_pending = {
|
|
2564
|
+
task
|
|
2565
|
+
for name, task in tasks.items()
|
|
2566
|
+
if task in pending and name != "provider_disconnected"
|
|
2567
|
+
}
|
|
2568
|
+
for task in owned_pending:
|
|
2569
|
+
task.cancel()
|
|
2570
|
+
if owned_pending:
|
|
2571
|
+
await asyncio.gather(*owned_pending, return_exceptions=True)
|
|
2572
|
+
if not done:
|
|
2573
|
+
return "timeout"
|
|
2574
|
+
# A crashed monitor is also "done"; it must not count as its condition.
|
|
2575
|
+
completed: set[str] = set()
|
|
2576
|
+
monitor_failures: dict[str, BaseException] = {}
|
|
2577
|
+
for name, task in tasks.items():
|
|
2578
|
+
if task not in done or task.cancelled():
|
|
2579
|
+
continue
|
|
2580
|
+
exc = task.exception()
|
|
2581
|
+
if exc is None:
|
|
2582
|
+
completed.add(name)
|
|
2583
|
+
else:
|
|
2584
|
+
monitor_failures[name] = exc
|
|
2585
|
+
for name, exc in monitor_failures.items():
|
|
2586
|
+
logger.warning(
|
|
2587
|
+
"conversation end monitor failed",
|
|
2588
|
+
exc_info=redacted_exc_info(exc),
|
|
2589
|
+
extra={"monitor": name, "target_identity": target_identity},
|
|
2590
|
+
)
|
|
2591
|
+
# A bridge task error is still a real provider-side disconnect.
|
|
2592
|
+
if "provider_disconnected" in monitor_failures:
|
|
2593
|
+
completed.add("provider_disconnected")
|
|
2594
|
+
for reason in (
|
|
2595
|
+
"simulator_end_call",
|
|
2596
|
+
"target_disconnected",
|
|
2597
|
+
"room_disconnected",
|
|
2598
|
+
"no_conversation",
|
|
2599
|
+
# A farewell loop is a finished conversation, so it outranks the silence backstop
|
|
2600
|
+
# that would otherwise report the same call as a stall.
|
|
2601
|
+
"closing_loop",
|
|
2602
|
+
"conversation_silence_timeout",
|
|
2603
|
+
"conversation_stalled",
|
|
2604
|
+
"provider_disconnected",
|
|
2605
|
+
"closed",
|
|
2606
|
+
):
|
|
2607
|
+
if reason in completed:
|
|
2608
|
+
return "session_closed" if reason == "closed" else reason
|
|
2609
|
+
if monitor_failures:
|
|
2610
|
+
return "monitor_failed"
|
|
2611
|
+
return "session_closed"
|
|
2612
|
+
finally:
|
|
2613
|
+
_remove_room_listener(session, "close", on_close)
|
|
2614
|
+
_remove_room_listener(
|
|
2615
|
+
room,
|
|
2616
|
+
"participant_disconnected",
|
|
2617
|
+
on_participant_disconnected,
|
|
2618
|
+
)
|
|
2619
|
+
_remove_room_listener(room, "disconnected", on_room_disconnected)
|
|
2620
|
+
|
|
2621
|
+
|
|
2622
|
+
_CLOSING_PHRASES = (
|
|
2623
|
+
"goodbye",
|
|
2624
|
+
"bye",
|
|
2625
|
+
"take care",
|
|
2626
|
+
"have a great day",
|
|
2627
|
+
"have a good day",
|
|
2628
|
+
"have a wonderful day",
|
|
2629
|
+
"you too",
|
|
2630
|
+
)
|
|
2631
|
+
|
|
2632
|
+
_CLOSING_EXCHANGE_LIMIT = 4
|
|
2633
|
+
|
|
2634
|
+
|
|
2635
|
+
# Words a farewell is allowed to be made of. Anything outside this set is substance, whatever the
|
|
2636
|
+
# turn's length: "yes it is, bye" is an answer and ending on it would cut a live call short.
|
|
2637
|
+
_CLOSING_FILLER = frozenset(
|
|
2638
|
+
"""
|
|
2639
|
+
a again alright and bye byebye care cheers day drive evening fine good goodbye great
|
|
2640
|
+
have later lovely morning much nice night ok okay perfect right safe see so soon sounds
|
|
2641
|
+
speak sure take talk thank thanks then to tomorrow too well wonderful you your
|
|
2642
|
+
""".split()
|
|
2643
|
+
)
|
|
2644
|
+
|
|
2645
|
+
|
|
2646
|
+
def _is_closing_only(text: str) -> bool:
|
|
2647
|
+
"""Whether a turn is nothing but a farewell.
|
|
2648
|
+
|
|
2649
|
+
Deliberately narrow: a turn that closes AND carries anything else (a question, a fact, a
|
|
2650
|
+
correction) is still conversation, and ending on it would cut a live call short.
|
|
2651
|
+
|
|
2652
|
+
Decided on whether every word is farewell filler rather than on a word count. A cap of six
|
|
2653
|
+
words classified "Sounds great, thanks. Talk tomorrow. Bye." as a farewell and "Sounds good,
|
|
2654
|
+
talk to you then. Bye." as conversation, purely because the second has one more word, and the
|
|
2655
|
+
engine then asked the caller for two further turns and got two more goodbyes.
|
|
2656
|
+
"""
|
|
2657
|
+
stripped = "".join(
|
|
2658
|
+
character.lower() if character.isalnum() or character.isspace() else " "
|
|
2659
|
+
for character in (text or "")
|
|
2660
|
+
).split()
|
|
2661
|
+
if not stripped or len(stripped) > 12:
|
|
2662
|
+
return False
|
|
2663
|
+
joined = " ".join(stripped)
|
|
2664
|
+
if not any(phrase in joined for phrase in _CLOSING_PHRASES):
|
|
2665
|
+
return False
|
|
2666
|
+
return not (set(stripped) - _CLOSING_FILLER)
|
|
2667
|
+
|
|
2668
|
+
|
|
2669
|
+
def _stop_any_further_speech(session: Any) -> None:
|
|
2670
|
+
"""Cancel anything already in flight, so the farewell is the last thing said.
|
|
2671
|
+
|
|
2672
|
+
Noticing the farewell only stops us asking for the NEXT turn. A reply already being generated
|
|
2673
|
+
still plays, which is how "Take care." arrived after a correct goodbye on a measured call. The
|
|
2674
|
+
farewell itself is already in history, meaning its own audio finished, so there is nothing of
|
|
2675
|
+
the caller's left to cut off here.
|
|
2676
|
+
"""
|
|
2677
|
+
try:
|
|
2678
|
+
session.interrupt(force=True)
|
|
2679
|
+
except Exception: # noqa: BLE001 - nothing in flight, or a session already shutting down
|
|
2680
|
+
logger.debug("nothing to interrupt when the call was closed", exc_info=True)
|
|
2681
|
+
|
|
2682
|
+
|
|
2683
|
+
async def _wait_for_closing_loop(
|
|
2684
|
+
session: AgentSession,
|
|
2685
|
+
*,
|
|
2686
|
+
limit: int = _CLOSING_EXCHANGE_LIMIT,
|
|
2687
|
+
) -> None:
|
|
2688
|
+
"""Finish once both sides are only trading farewells.
|
|
2689
|
+
|
|
2690
|
+
A simulator that does not reach for ``endCall`` leaves the target answering goodbye with
|
|
2691
|
+
goodbye until the deadline. One such call ran seventy-six turns, held its worker past the
|
|
2692
|
+
world pool's patience and cost the rest of that job its worlds, so this ends the call on the
|
|
2693
|
+
evidence already in the transcript rather than waiting for a timeout that arrives too late.
|
|
2694
|
+
"""
|
|
2695
|
+
while True:
|
|
2696
|
+
messages = _session_messages(session)
|
|
2697
|
+
spoken = [
|
|
2698
|
+
message for message in messages if (message.get("content") or "").strip()
|
|
2699
|
+
]
|
|
2700
|
+
# The caller's own farewell is the end of the call from its side, so there is no reason to
|
|
2701
|
+
# ask it for another turn. Waiting for a loop of farewells is what produced "Talk
|
|
2702
|
+
# tomorrow. Bye." followed by "Take care." and then "Bye." -- three closings where the
|
|
2703
|
+
# first was already correct. Rule 10 of the caller's prompt says exactly this, and an
|
|
2704
|
+
# instruction cannot enforce it: the model only speaks again because it was asked to.
|
|
2705
|
+
if (
|
|
2706
|
+
_turns_from_each_side(spoken) >= 1
|
|
2707
|
+
and spoken
|
|
2708
|
+
and spoken[-1].get("role") == _CALLER
|
|
2709
|
+
and _is_closing_only(str(spoken[-1].get("content") or ""))
|
|
2710
|
+
):
|
|
2711
|
+
logger.info("the caller said goodbye, ending the call")
|
|
2712
|
+
_stop_any_further_speech(session)
|
|
2713
|
+
return
|
|
2714
|
+
tail = spoken[-limit:]
|
|
2715
|
+
if len(tail) == limit and all(
|
|
2716
|
+
_is_closing_only(str(message.get("content") or "")) for message in tail
|
|
2717
|
+
):
|
|
2718
|
+
logger.info(
|
|
2719
|
+
"closing loop: last %d turns were farewells only, ending the call",
|
|
2720
|
+
limit,
|
|
2721
|
+
)
|
|
2722
|
+
_stop_any_further_speech(session)
|
|
2723
|
+
return
|
|
2724
|
+
# A turn lands in history only after its TTS finishes, so every poll interval between the
|
|
2725
|
+
# farewell committing and this noticing is time in which the caller can be asked for
|
|
2726
|
+
# another turn. Measured: one trailing turn survived at a one-second poll.
|
|
2727
|
+
await asyncio.sleep(0.25)
|
|
2728
|
+
|
|
2729
|
+
|
|
2730
|
+
async def _wait_for_conversation_silence(
|
|
2731
|
+
session: AgentSession,
|
|
2732
|
+
*,
|
|
2733
|
+
quiet_seconds: float = _SILENCE_BACKSTOP_SECONDS,
|
|
2734
|
+
min_turn_messages: int = 0,
|
|
2735
|
+
) -> None:
|
|
2736
|
+
"""Finish only after a long, genuine stretch of mutual silence.
|
|
2737
|
+
|
|
2738
|
+
The simulator ends a call by calling ``endCall`` once the scenario is done;
|
|
2739
|
+
this is only the backstop for a conversation that has actually stalled (or
|
|
2740
|
+
already finished but never hung up). It deliberately does **not** look at the
|
|
2741
|
+
message count — a run is never cut off at a floor, it runs as long as turns
|
|
2742
|
+
keep flowing. The timer resets on every new message and while either side is
|
|
2743
|
+
speaking, so only a real ``quiet_seconds`` gap of nothing ends the call.
|
|
2744
|
+
|
|
2745
|
+
Parks until the first non-empty turn: a call where nobody ever spoke is the
|
|
2746
|
+
``no_conversation`` monitor's condition, and this backstop firing first
|
|
2747
|
+
mislabeled dead calls as merely settled.
|
|
2748
|
+
"""
|
|
2749
|
+
last_signature: tuple[tuple[str, str], ...] | None = None
|
|
2750
|
+
stable_since: float | None = None
|
|
2751
|
+
loop = asyncio.get_running_loop()
|
|
2752
|
+
while True:
|
|
2753
|
+
messages = _session_messages(session)
|
|
2754
|
+
signature = tuple((message["role"], message["content"]) for message in messages)
|
|
2755
|
+
if not any(message["content"] for message in messages):
|
|
2756
|
+
last_signature = signature
|
|
2757
|
+
stable_since = None
|
|
2758
|
+
await asyncio.sleep(0.1)
|
|
2759
|
+
continue
|
|
2760
|
+
now = loop.time()
|
|
2761
|
+
participant_busy = _either_side_busy(session)
|
|
2762
|
+
floor, _ = _turn_requirements(min_turn_messages)
|
|
2763
|
+
# Far enough in for the measured window to beat the fixed one. A third of the floor is a
|
|
2764
|
+
# threshold, not a derived figure: enough turns to have timed a reply, well short of done.
|
|
2765
|
+
settled = min_turn_messages > 0 and _turns_from_each_side(messages) >= max(
|
|
2766
|
+
2, floor // 3
|
|
2767
|
+
)
|
|
2768
|
+
effective_quiet = (
|
|
2769
|
+
_settled_silence_window(
|
|
2770
|
+
_observed_agent_reply_seconds(messages), quiet_seconds
|
|
2771
|
+
)
|
|
2772
|
+
if settled
|
|
2773
|
+
else quiet_seconds
|
|
2774
|
+
)
|
|
2775
|
+
if participant_busy:
|
|
2776
|
+
stable_since = None
|
|
2777
|
+
elif stable_since is None or signature != last_signature:
|
|
2778
|
+
stable_since = now
|
|
2779
|
+
elif now - stable_since >= effective_quiet:
|
|
2780
|
+
return
|
|
2781
|
+
last_signature = signature
|
|
2782
|
+
await asyncio.sleep(0.1)
|
|
2783
|
+
|
|
2784
|
+
|
|
2785
|
+
async def _wait_for_conversation_never_started(
|
|
2786
|
+
session: AgentSession,
|
|
2787
|
+
*,
|
|
2788
|
+
timeout_seconds: float,
|
|
2789
|
+
) -> None:
|
|
2790
|
+
"""Completes only when no non-empty turn has ever been committed; parks
|
|
2791
|
+
forever (until cancelled) once the conversation has actually started."""
|
|
2792
|
+
loop = asyncio.get_running_loop()
|
|
2793
|
+
deadline = loop.time() + timeout_seconds
|
|
2794
|
+
while loop.time() < deadline:
|
|
2795
|
+
if any(message["content"] for message in _session_messages(session)):
|
|
2796
|
+
await asyncio.Event().wait()
|
|
2797
|
+
await asyncio.sleep(0.5)
|
|
2798
|
+
|
|
2799
|
+
|
|
2800
|
+
def _voicemail_tone_style() -> str:
|
|
2801
|
+
"""The style whose tone this call plays; empty for a person, and for a full mailbox by design."""
|
|
2802
|
+
if not _answered_by_voicemail():
|
|
2803
|
+
return ""
|
|
2804
|
+
style = (
|
|
2805
|
+
os.environ.get("HARNESS_VOICEMAIL_STYLE", "").strip().lower()
|
|
2806
|
+
or _DEFAULT_VOICEMAIL_STYLE
|
|
2807
|
+
)
|
|
2808
|
+
return style if style in _VOICEMAIL_TONE_BY_STYLE else ""
|
|
2809
|
+
|
|
2810
|
+
|
|
2811
|
+
def _downloaded_audio(source: str) -> str | None:
|
|
2812
|
+
"""A local copy of a remote audio file, or None: a call heard in the clear beats a dropped one."""
|
|
2813
|
+
import tempfile
|
|
2814
|
+
import urllib.request
|
|
2815
|
+
|
|
2816
|
+
try:
|
|
2817
|
+
suffix = ".mp3" if ".mp3" in source else ".ogg" if ".ogg" in source else ".wav"
|
|
2818
|
+
with urllib.request.urlopen(source, timeout=15) as response:
|
|
2819
|
+
data = response.read()
|
|
2820
|
+
handle = tempfile.NamedTemporaryFile(delete=False, suffix=suffix)
|
|
2821
|
+
handle.write(data)
|
|
2822
|
+
handle.close()
|
|
2823
|
+
return handle.name
|
|
2824
|
+
except Exception:
|
|
2825
|
+
return None
|
|
2826
|
+
|
|
2827
|
+
|
|
2828
|
+
def _frame_at_mixer_rate(frame: "rtc.AudioFrame") -> "rtc.AudioFrame":
|
|
2829
|
+
"""The same audio at the mixer's rate, which reinterprets rather than resamples what it is given."""
|
|
2830
|
+
if frame.sample_rate == _BACKGROUND_MIXER_RATE:
|
|
2831
|
+
return frame
|
|
2832
|
+
resampler = _MIXER_RESAMPLERS.get((frame.sample_rate, frame.num_channels))
|
|
2833
|
+
if resampler is None:
|
|
2834
|
+
resampler = PCMResampler(
|
|
2835
|
+
from_rate=frame.sample_rate,
|
|
2836
|
+
to_rate=_BACKGROUND_MIXER_RATE,
|
|
2837
|
+
channels=frame.num_channels,
|
|
2838
|
+
)
|
|
2839
|
+
_MIXER_RESAMPLERS[(frame.sample_rate, frame.num_channels)] = resampler
|
|
2840
|
+
converted = resampler.convert(bytes(frame.data))
|
|
2841
|
+
return rtc.AudioFrame(
|
|
2842
|
+
data=converted,
|
|
2843
|
+
sample_rate=_BACKGROUND_MIXER_RATE,
|
|
2844
|
+
num_channels=frame.num_channels,
|
|
2845
|
+
samples_per_channel=len(converted) // (2 * frame.num_channels),
|
|
2846
|
+
)
|
|
2847
|
+
|
|
2848
|
+
|
|
2849
|
+
def _tone_frame(hz: float, seconds: float) -> "rtc.AudioFrame":
|
|
2850
|
+
"""One frame of sine, faded in and out: a burst at full amplitude clicks and a detector hears the click."""
|
|
2851
|
+
total = int(_BACKGROUND_MIXER_RATE * seconds)
|
|
2852
|
+
fade = max(1, int(_BACKGROUND_MIXER_RATE * 0.01))
|
|
2853
|
+
samples = array.array("h")
|
|
2854
|
+
for index in range(total):
|
|
2855
|
+
gain = min(1.0, index / fade, (total - index) / fade)
|
|
2856
|
+
samples.append(
|
|
2857
|
+
int(
|
|
2858
|
+
32767
|
|
2859
|
+
* 0.9
|
|
2860
|
+
* gain
|
|
2861
|
+
* math.sin(2 * math.pi * hz * index / _BACKGROUND_MIXER_RATE)
|
|
2862
|
+
)
|
|
2863
|
+
)
|
|
2864
|
+
return rtc.AudioFrame(
|
|
2865
|
+
data=samples.tobytes(),
|
|
2866
|
+
sample_rate=_BACKGROUND_MIXER_RATE,
|
|
2867
|
+
num_channels=1,
|
|
2868
|
+
samples_per_channel=total,
|
|
2869
|
+
)
|
|
2870
|
+
|
|
2871
|
+
|
|
2872
|
+
def _answered_by_voicemail() -> bool:
|
|
2873
|
+
"""Whether a mailbox answered rather than a person; set per scenario by the call runner."""
|
|
2874
|
+
return os.environ.get("HARNESS_ANSWERED_BY", "").strip().lower() == "voicemail"
|
|
2875
|
+
|
|
2876
|
+
|
|
2877
|
+
async def _open_if_nobody_speaks_first(
|
|
2878
|
+
session: AgentSession,
|
|
2879
|
+
customer_agent: Any,
|
|
2880
|
+
*,
|
|
2881
|
+
timeout_seconds: float,
|
|
2882
|
+
) -> None:
|
|
2883
|
+
"""Have the simulated person open the conversation when the other side never does.
|
|
2884
|
+
|
|
2885
|
+
Only for a call the agent was supposed to start. It opens exactly the way a simulator-first call
|
|
2886
|
+
does, through ``open_conversation``, so the person's own initial message is used where the
|
|
2887
|
+
persona has one. Returns as soon as anybody speaks, which is the ordinary case.
|
|
2888
|
+
"""
|
|
2889
|
+
loop = asyncio.get_running_loop()
|
|
2890
|
+
deadline = loop.time() + timeout_seconds
|
|
2891
|
+
while loop.time() < deadline:
|
|
2892
|
+
if any(message["content"] for message in _session_messages(session)):
|
|
2893
|
+
return
|
|
2894
|
+
await asyncio.sleep(0.2)
|
|
2895
|
+
if any(message["content"] for message in _session_messages(session)):
|
|
2896
|
+
return
|
|
2897
|
+
logger.warning(
|
|
2898
|
+
"no first turn after %ss; the simulated person opens instead", timeout_seconds
|
|
2899
|
+
)
|
|
2900
|
+
try:
|
|
2901
|
+
customer_agent.open_conversation()
|
|
2902
|
+
except Exception: # noqa: BLE001 - a call that cannot be opened is the case's own failure
|
|
2903
|
+
logger.warning("the simulated person could not open the call", exc_info=True)
|
|
2904
|
+
|
|
2905
|
+
|
|
2906
|
+
async def _wait_for_agent_first_silence(
|
|
2907
|
+
session: AgentSession,
|
|
2908
|
+
*,
|
|
2909
|
+
timeout_seconds: float,
|
|
2910
|
+
) -> None:
|
|
2911
|
+
last_signature: tuple[tuple[str, str], ...] = ()
|
|
2912
|
+
last_change = asyncio.get_running_loop().time()
|
|
2913
|
+
while True:
|
|
2914
|
+
messages = _session_messages(session)
|
|
2915
|
+
signature = tuple((message["role"], message["content"]) for message in messages)
|
|
2916
|
+
# A turn lands in history only after its TTS finishes, so an in-flight utterance longer
|
|
2917
|
+
# than the timeout must count as activity -- and so must an agent that is still thinking,
|
|
2918
|
+
# or the caller opens over the top of a reply that was on its way.
|
|
2919
|
+
if signature != last_signature or _either_side_busy(session):
|
|
2920
|
+
last_signature = signature
|
|
2921
|
+
last_change = asyncio.get_running_loop().time()
|
|
2922
|
+
roles = {message["role"] for message in messages if message["content"]}
|
|
2923
|
+
if {"user", "assistant"}.issubset(
|
|
2924
|
+
roles
|
|
2925
|
+
) and asyncio.get_running_loop().time() - last_change >= timeout_seconds:
|
|
2926
|
+
return
|
|
2927
|
+
await asyncio.sleep(0.1)
|
|
2928
|
+
|
|
2929
|
+
|
|
2930
|
+
def _session_messages(session: AgentSession) -> list[dict[str, Any]]:
|
|
2931
|
+
"""Return normalized transcript messages with real per-item speech timing.
|
|
2932
|
+
|
|
2933
|
+
Each dict carries:
|
|
2934
|
+
role, content: str
|
|
2935
|
+
started_speaking_at, stopped_speaking_at: float | None
|
|
2936
|
+
Real audio timing from ``ChatMessage.metrics`` (seconds since epoch).
|
|
2937
|
+
See livekit.agents.llm.chat_context.MetricsReport.
|
|
2938
|
+
created_at: float
|
|
2939
|
+
Fallback wall-clock stamp from ``ChatMessage.created_at`` (used when
|
|
2940
|
+
the metrics timestamps are missing, e.g. text-only turns).
|
|
2941
|
+
interrupted: bool
|
|
2942
|
+
e2e_latency: float | None
|
|
2943
|
+
Agent-side turn latency, when reported by LiveKit.
|
|
2944
|
+
|
|
2945
|
+
Downstream code turns these into millisecond offsets so the platform can
|
|
2946
|
+
recompute WPM, talk-ratio and interruption counts with real overlap data.
|
|
2947
|
+
"""
|
|
2948
|
+
messages: list[dict[str, Any]] = []
|
|
2949
|
+
for item in session.history.items:
|
|
2950
|
+
if getattr(item, "type", None) != "message":
|
|
2951
|
+
continue
|
|
2952
|
+
role = getattr(item, "role", None)
|
|
2953
|
+
text = getattr(item, "text_content", None)
|
|
2954
|
+
if role is None or text is None:
|
|
2955
|
+
continue
|
|
2956
|
+
interrupted = bool(getattr(item, "interrupted", False))
|
|
2957
|
+
created_at = float(getattr(item, "created_at", 0.0) or 0.0)
|
|
2958
|
+
metrics = getattr(item, "metrics", None) or {}
|
|
2959
|
+
started_speaking_at = _maybe_float(metrics.get("started_speaking_at"))
|
|
2960
|
+
stopped_speaking_at = _maybe_float(metrics.get("stopped_speaking_at"))
|
|
2961
|
+
e2e_latency = _maybe_float(metrics.get("e2e_latency"))
|
|
2962
|
+
current: dict[str, Any] = {
|
|
2963
|
+
"role": str(role),
|
|
2964
|
+
"content": str(text),
|
|
2965
|
+
"created_at": created_at,
|
|
2966
|
+
"started_speaking_at": started_speaking_at,
|
|
2967
|
+
"stopped_speaking_at": stopped_speaking_at,
|
|
2968
|
+
"interrupted": interrupted,
|
|
2969
|
+
"e2e_latency": e2e_latency,
|
|
2970
|
+
}
|
|
2971
|
+
if messages and messages[-1]["role"] == current["role"]:
|
|
2972
|
+
previous = messages[-1]
|
|
2973
|
+
previous_text = previous["content"]
|
|
2974
|
+
if current["content"].startswith(previous_text):
|
|
2975
|
+
# Newer emission extends the previous partial — keep the
|
|
2976
|
+
# earliest start we saw, adopt the latest stop.
|
|
2977
|
+
current["started_speaking_at"] = (
|
|
2978
|
+
previous.get("started_speaking_at")
|
|
2979
|
+
or current["started_speaking_at"]
|
|
2980
|
+
)
|
|
2981
|
+
current["created_at"] = previous["created_at"] or created_at
|
|
2982
|
+
messages[-1] = current
|
|
2983
|
+
elif previous_text.startswith(current["content"]):
|
|
2984
|
+
previous["interrupted"] = previous.get("interrupted") or interrupted
|
|
2985
|
+
previous["stopped_speaking_at"] = (
|
|
2986
|
+
previous.get("stopped_speaking_at")
|
|
2987
|
+
or current["stopped_speaking_at"]
|
|
2988
|
+
)
|
|
2989
|
+
elif previous.get("interrupted") or interrupted:
|
|
2990
|
+
previous["content"] = f"{previous_text} {current['content']}".strip()
|
|
2991
|
+
previous["interrupted"] = interrupted
|
|
2992
|
+
previous["stopped_speaking_at"] = current[
|
|
2993
|
+
"stopped_speaking_at"
|
|
2994
|
+
] or previous.get("stopped_speaking_at")
|
|
2995
|
+
else:
|
|
2996
|
+
messages.append(current)
|
|
2997
|
+
continue
|
|
2998
|
+
messages.append(current)
|
|
2999
|
+
return messages
|
|
3000
|
+
|
|
3001
|
+
|
|
3002
|
+
def _maybe_float(value: Any) -> float | None:
|
|
3003
|
+
if value is None:
|
|
3004
|
+
return None
|
|
3005
|
+
try:
|
|
3006
|
+
return float(value)
|
|
3007
|
+
except (TypeError, ValueError):
|
|
3008
|
+
return None
|
|
3009
|
+
|
|
3010
|
+
|
|
3011
|
+
def _canonical_report_messages(session: AgentSession) -> list[dict[str, Any]]:
|
|
3012
|
+
"""Emit report messages with roles remapped to the test-agent perspective.
|
|
3013
|
+
|
|
3014
|
+
LiveKit reports our simulator as ``assistant`` and the target agent as
|
|
3015
|
+
``user`` (the SDK connects with role ``agent``); we swap those so the
|
|
3016
|
+
downstream platform sees:
|
|
3017
|
+
role="user" → simulator / customer
|
|
3018
|
+
role="assistant" → agent-under-test
|
|
3019
|
+
which matches the CallTranscript convention.
|
|
3020
|
+
|
|
3021
|
+
Timing anchors (``started_speaking_at`` / ``stopped_speaking_at``) travel
|
|
3022
|
+
through unchanged so the platform can derive ms offsets.
|
|
3023
|
+
"""
|
|
3024
|
+
role_map = {"assistant": "user", "user": "assistant"}
|
|
3025
|
+
messages: list[dict[str, Any]] = []
|
|
3026
|
+
for source in _session_messages(session):
|
|
3027
|
+
messages.append(
|
|
3028
|
+
{
|
|
3029
|
+
"role": role_map.get(source["role"], source["role"]),
|
|
3030
|
+
"content": source["content"],
|
|
3031
|
+
"created_at": source.get("created_at"),
|
|
3032
|
+
"started_speaking_at": source.get("started_speaking_at"),
|
|
3033
|
+
"stopped_speaking_at": source.get("stopped_speaking_at"),
|
|
3034
|
+
"interrupted": source.get("interrupted", False),
|
|
3035
|
+
"e2e_latency": source.get("e2e_latency"),
|
|
3036
|
+
}
|
|
3037
|
+
)
|
|
3038
|
+
return messages
|
|
3039
|
+
|
|
3040
|
+
|
|
3041
|
+
def _merge_captured_target_turns(
|
|
3042
|
+
messages: list[dict[str, Any]],
|
|
3043
|
+
captured_target_turns: list[dict[str, Any]] | None,
|
|
3044
|
+
) -> list[dict[str, Any]]:
|
|
3045
|
+
"""Restore native target-turn timing, and append any target turn that never
|
|
3046
|
+
reached the session history.
|
|
3047
|
+
|
|
3048
|
+
Native target turns are fed to the simulator via ``generate_reply(
|
|
3049
|
+
user_input=text)`` — a text input with no audio metrics — so their report
|
|
3050
|
+
entries carry a start but no ``stopped_speaking_at``: zero-duration turns
|
|
3051
|
+
that leave bot WPM, latency, and talk-ratio unpopulated. Each captured turn
|
|
3052
|
+
carries receiver-side wall-clock timing (see ``_forward_target_transcription``);
|
|
3053
|
+
here we (a) patch it onto the matching ``assistant`` turns missing a real
|
|
3054
|
+
stop, and (b) append the trailing turn delivered after the simulator drained
|
|
3055
|
+
("speech scheduling is paused"). One turn can arrive as several partial or
|
|
3056
|
+
extended emissions, so match by containment and aggregate min-start/max-stop.
|
|
3057
|
+
Only populated by the native transcription handler — VAPI/Retell are untouched.
|
|
3058
|
+
"""
|
|
3059
|
+
if not captured_target_turns:
|
|
3060
|
+
return messages
|
|
3061
|
+
|
|
3062
|
+
def _matching(text: str) -> list[dict[str, Any]]:
|
|
3063
|
+
result = []
|
|
3064
|
+
for captured in captured_target_turns:
|
|
3065
|
+
cap_text = (captured.get("content") or "").strip()
|
|
3066
|
+
if cap_text and (text in cap_text or cap_text in text):
|
|
3067
|
+
result.append(captured)
|
|
3068
|
+
return result
|
|
3069
|
+
|
|
3070
|
+
# (a) Fill timing onto existing assistant turns that lack a real stop; never
|
|
3071
|
+
# override genuine audio metrics if livekit-agents ever populates them.
|
|
3072
|
+
for message in messages:
|
|
3073
|
+
if message.get("role") != "assistant":
|
|
3074
|
+
continue
|
|
3075
|
+
text = (message.get("content") or "").strip()
|
|
3076
|
+
if not text:
|
|
3077
|
+
continue
|
|
3078
|
+
started = message.get("started_speaking_at")
|
|
3079
|
+
stopped = message.get("stopped_speaking_at")
|
|
3080
|
+
if (
|
|
3081
|
+
isinstance(started, (int, float))
|
|
3082
|
+
and isinstance(stopped, (int, float))
|
|
3083
|
+
and stopped > started
|
|
3084
|
+
):
|
|
3085
|
+
continue
|
|
3086
|
+
matched = _matching(text)
|
|
3087
|
+
starts = [
|
|
3088
|
+
c["started_speaking_at"]
|
|
3089
|
+
for c in matched
|
|
3090
|
+
if isinstance(c.get("started_speaking_at"), (int, float))
|
|
3091
|
+
]
|
|
3092
|
+
stops = [
|
|
3093
|
+
c["stopped_speaking_at"]
|
|
3094
|
+
for c in matched
|
|
3095
|
+
if isinstance(c.get("stopped_speaking_at"), (int, float))
|
|
3096
|
+
]
|
|
3097
|
+
if starts:
|
|
3098
|
+
message["started_speaking_at"] = min(starts)
|
|
3099
|
+
if not isinstance(message.get("created_at"), (int, float)):
|
|
3100
|
+
message["created_at"] = min(starts)
|
|
3101
|
+
if stops:
|
|
3102
|
+
message["stopped_speaking_at"] = max(stops)
|
|
3103
|
+
|
|
3104
|
+
# (b) Append target turns that never reached the report at all.
|
|
3105
|
+
assistant_texts = [
|
|
3106
|
+
(m.get("content") or "").strip()
|
|
3107
|
+
for m in messages
|
|
3108
|
+
if m.get("role") == "assistant" and m.get("content")
|
|
3109
|
+
]
|
|
3110
|
+
|
|
3111
|
+
def _already_present(text: str) -> bool:
|
|
3112
|
+
return any(text in existing or existing in text for existing in assistant_texts)
|
|
3113
|
+
|
|
3114
|
+
last_ts = 0.0
|
|
3115
|
+
for m in messages:
|
|
3116
|
+
for key in ("stopped_speaking_at", "started_speaking_at", "created_at"):
|
|
3117
|
+
value = m.get(key)
|
|
3118
|
+
if isinstance(value, (int, float)) and value > last_ts:
|
|
3119
|
+
last_ts = value
|
|
3120
|
+
|
|
3121
|
+
merged = list(messages)
|
|
3122
|
+
for offset, captured in enumerate(captured_target_turns, start=1):
|
|
3123
|
+
text = (captured.get("content") or "").strip()
|
|
3124
|
+
if not text or _already_present(text):
|
|
3125
|
+
continue
|
|
3126
|
+
started = captured.get("started_speaking_at")
|
|
3127
|
+
if not isinstance(started, (int, float)):
|
|
3128
|
+
started = (last_ts + offset) if last_ts else None
|
|
3129
|
+
stopped = captured.get("stopped_speaking_at")
|
|
3130
|
+
if not isinstance(stopped, (int, float)):
|
|
3131
|
+
stopped = started
|
|
3132
|
+
merged.append(
|
|
3133
|
+
{
|
|
3134
|
+
"role": "assistant",
|
|
3135
|
+
"content": text,
|
|
3136
|
+
"created_at": started,
|
|
3137
|
+
"started_speaking_at": started,
|
|
3138
|
+
"stopped_speaking_at": stopped,
|
|
3139
|
+
"interrupted": False,
|
|
3140
|
+
"e2e_latency": None,
|
|
3141
|
+
}
|
|
3142
|
+
)
|
|
3143
|
+
assistant_texts.append(text)
|
|
3144
|
+
return merged
|
|
3145
|
+
|
|
3146
|
+
|
|
3147
|
+
def _caller_never_spoke(messages: list[dict[str, Any]]) -> bool:
|
|
3148
|
+
"""Whether the simulated caller's turns exist as text with no audio behind them.
|
|
3149
|
+
|
|
3150
|
+
Speech synthesis that fails still leaves the caller's line in the transcript, so a mute
|
|
3151
|
+
simulator and a silent agent produce the same stall unless the missing audio is read directly.
|
|
3152
|
+
"""
|
|
3153
|
+
|
|
3154
|
+
def timed(message: dict[str, Any]) -> bool:
|
|
3155
|
+
return isinstance(message.get("started_speaking_at"), (int, float))
|
|
3156
|
+
|
|
3157
|
+
spoken = [
|
|
3158
|
+
message
|
|
3159
|
+
for message in messages
|
|
3160
|
+
if message.get("role") == "user" and (message.get("content") or "").strip()
|
|
3161
|
+
]
|
|
3162
|
+
if not spoken or any(timed(message) for message in spoken):
|
|
3163
|
+
return False
|
|
3164
|
+
# Only the agent's turns carrying timing makes the caller's missing timing evidence of
|
|
3165
|
+
# silence rather than a transcript that simply does not record when anyone spoke.
|
|
3166
|
+
return any(
|
|
3167
|
+
timed(message) for message in messages if message.get("role") == "assistant"
|
|
3168
|
+
)
|
|
3169
|
+
|
|
3170
|
+
|
|
3171
|
+
def _recover_successful_provider_end_call(
|
|
3172
|
+
outcome: _CaseOutcome,
|
|
3173
|
+
provider_summary: EvidenceSourceSummary,
|
|
3174
|
+
) -> None:
|
|
3175
|
+
"""Keep a clean provider hangup separate from scenario correctness.
|
|
3176
|
+
|
|
3177
|
+
Retell can disconnect immediately after its ``end_call`` tool succeeds, before the final
|
|
3178
|
+
synthesized farewell is committed into LiveKit's transcript. A minimum-turn guard may have
|
|
3179
|
+
provisionally classified that as an incomplete conversation. Provider evidence is the
|
|
3180
|
+
authoritative lifecycle signal here: promote the call to completed and let scenario checks
|
|
3181
|
+
report any business-goal failure. Requiring both roles protects genuine mute/no-conversation
|
|
3182
|
+
failures from being hidden by a malformed provider trace.
|
|
3183
|
+
"""
|
|
3184
|
+
if outcome.failure is None or outcome.failure.code not in {
|
|
3185
|
+
"insufficient_conversation",
|
|
3186
|
+
"target_disconnected",
|
|
3187
|
+
"room_disconnected",
|
|
3188
|
+
}:
|
|
3189
|
+
return
|
|
3190
|
+
if not _has_role_alternation(outcome.messages):
|
|
3191
|
+
return
|
|
3192
|
+
calls = provider_summary.metadata.get("tool_calls")
|
|
3193
|
+
if not isinstance(calls, list):
|
|
3194
|
+
return
|
|
3195
|
+
ended_cleanly = any(
|
|
3196
|
+
isinstance(call, dict)
|
|
3197
|
+
and str(call.get("name") or "").strip().lower() == "end_call"
|
|
3198
|
+
and call.get("ok") is not False
|
|
3199
|
+
for call in calls
|
|
3200
|
+
)
|
|
3201
|
+
if not ended_cleanly:
|
|
3202
|
+
return
|
|
3203
|
+
outcome.status = TestCaseStatus.COMPLETED
|
|
3204
|
+
outcome.failure = None
|
|
3205
|
+
outcome.metadata["provider_end_call_recovered"] = True
|
|
3206
|
+
|
|
3207
|
+
|
|
3208
|
+
def _reconcile_provider_observation(
|
|
3209
|
+
outcome: _CaseOutcome,
|
|
3210
|
+
provider_summary: EvidenceSourceSummary,
|
|
3211
|
+
) -> None:
|
|
3212
|
+
"""Recover provider-native speech and deterministic target tool failures."""
|
|
3213
|
+
raw_messages = provider_summary.metadata.get("messages")
|
|
3214
|
+
provider_messages: list[dict[str, Any]] = []
|
|
3215
|
+
if isinstance(raw_messages, list):
|
|
3216
|
+
for raw in raw_messages:
|
|
3217
|
+
if not isinstance(raw, dict):
|
|
3218
|
+
continue
|
|
3219
|
+
role = str(raw.get("role") or "").strip().lower()
|
|
3220
|
+
content = str(raw.get("content") or "").strip()
|
|
3221
|
+
if role not in {"user", "assistant"} or not content:
|
|
3222
|
+
continue
|
|
3223
|
+
provider_messages.append(dict(raw))
|
|
3224
|
+
if provider_messages and not outcome.messages:
|
|
3225
|
+
outcome.messages = provider_messages
|
|
3226
|
+
outcome.transcript = "\n".join(
|
|
3227
|
+
f"{message['role']}: {message['content']}" for message in provider_messages
|
|
3228
|
+
)
|
|
3229
|
+
outcome.metadata["provider_transcript_recovered"] = True
|
|
3230
|
+
|
|
3231
|
+
calls = provider_summary.metadata.get("tool_calls")
|
|
3232
|
+
failed_call = (
|
|
3233
|
+
next(
|
|
3234
|
+
(
|
|
3235
|
+
call
|
|
3236
|
+
for call in calls
|
|
3237
|
+
if isinstance(call, dict)
|
|
3238
|
+
and call.get("ok") is False
|
|
3239
|
+
and str(call.get("type") or "").lower() != "end_call"
|
|
3240
|
+
),
|
|
3241
|
+
None,
|
|
3242
|
+
)
|
|
3243
|
+
if isinstance(calls, list)
|
|
3244
|
+
else None
|
|
3245
|
+
)
|
|
3246
|
+
if failed_call is None or outcome.failure is None:
|
|
3247
|
+
return
|
|
3248
|
+
if outcome.failure.code not in {
|
|
3249
|
+
"insufficient_conversation",
|
|
3250
|
+
"target_disconnected",
|
|
3251
|
+
"room_disconnected",
|
|
3252
|
+
"no_conversation",
|
|
3253
|
+
"conversation_stalled",
|
|
3254
|
+
"conversation_silence_timeout",
|
|
3255
|
+
}:
|
|
3256
|
+
return
|
|
3257
|
+
name = str(failed_call.get("name") or "unknown")
|
|
3258
|
+
error = str(failed_call.get("error") or "tool call failed")
|
|
3259
|
+
if len(error) > 500:
|
|
3260
|
+
error = error[:500]
|
|
3261
|
+
outcome.status = TestCaseStatus.FAILED
|
|
3262
|
+
outcome.failure = SimulationFailure(
|
|
3263
|
+
stage=FailureStage.RUNNING,
|
|
3264
|
+
code="target_agent_tool_failed",
|
|
3265
|
+
message=f"Target agent tool {name!r} failed: {error}",
|
|
3266
|
+
retryable=False,
|
|
3267
|
+
provider=str(provider_summary.metadata.get("provider") or "provider"),
|
|
3268
|
+
details={
|
|
3269
|
+
"tool_name": name,
|
|
3270
|
+
"provider_end_reason": str(
|
|
3271
|
+
provider_summary.metadata.get("end_reason") or ""
|
|
3272
|
+
),
|
|
3273
|
+
},
|
|
3274
|
+
)
|
|
3275
|
+
outcome.metadata["provider_tool_failure_attributed"] = True
|
|
3276
|
+
|
|
3277
|
+
|
|
3278
|
+
def _turn_requirements(min_turn_messages: int) -> tuple[int, bool]:
|
|
3279
|
+
"""The turn floor and whether alternation is required: a mailbox is held to its greeting alone."""
|
|
3280
|
+
if _answered_by_voicemail():
|
|
3281
|
+
return min(min_turn_messages, _VOICEMAIL_MIN_TURN_MESSAGES), False
|
|
3282
|
+
return min_turn_messages, True
|
|
3283
|
+
|
|
3284
|
+
|
|
3285
|
+
def _has_role_alternation(messages: list[dict[str, Any]]) -> bool:
|
|
3286
|
+
roles = {msg.get("role") for msg in messages if msg.get("content")}
|
|
3287
|
+
return "user" in roles and "assistant" in roles
|
|
3288
|
+
|
|
3289
|
+
|
|
3290
|
+
def _turns_from_each_side(messages: list[dict[str, Any]]) -> int:
|
|
3291
|
+
"""How many turns the quieter speaker took; a total is inflated by one side's own filler."""
|
|
3292
|
+
spoken = [msg for msg in messages if msg.get("content")]
|
|
3293
|
+
return min(
|
|
3294
|
+
sum(1 for msg in spoken if msg.get("role") == "assistant"),
|
|
3295
|
+
sum(1 for msg in spoken if msg.get("role") == "user"),
|
|
3296
|
+
)
|
|
3297
|
+
|
|
3298
|
+
|
|
3299
|
+
# Two unanswered turns: one can be the caller finishing a thought, two means nobody is replying.
|
|
3300
|
+
_QUIET_AFTER_UNANSWERED_TURNS = 2
|
|
3301
|
+
|
|
3302
|
+
|
|
3303
|
+
def _target_has_gone_quiet(messages: list[dict[str, Any]]) -> bool:
|
|
3304
|
+
"""Whether the agent has stopped replying, so the floor can never be reached honestly.
|
|
3305
|
+
|
|
3306
|
+
The floor counts messages, and the caller's own turns count toward it, so a caller that is
|
|
3307
|
+
refused the tool talks to fill the silence and eventually buys its own permission. That is the
|
|
3308
|
+
opposite of what the floor is for. When the agent has spoken and then stopped, the caller is
|
|
3309
|
+
allowed to hang up instead.
|
|
3310
|
+
"""
|
|
3311
|
+
spoken = [message for message in messages if message.get("content")]
|
|
3312
|
+
if not any(message.get("role") == "assistant" for message in spoken):
|
|
3313
|
+
return False
|
|
3314
|
+
trailing = 0
|
|
3315
|
+
for message in reversed(spoken):
|
|
3316
|
+
if message.get("role") != "user":
|
|
3317
|
+
break
|
|
3318
|
+
trailing += 1
|
|
3319
|
+
return trailing >= _QUIET_AFTER_UNANSWERED_TURNS
|
|
3320
|
+
|
|
3321
|
+
|
|
3322
|
+
def _conversation_outcome(
|
|
3323
|
+
stop_reason: str,
|
|
3324
|
+
messages: list[dict[str, str]],
|
|
3325
|
+
*,
|
|
3326
|
+
min_turn_messages: int,
|
|
3327
|
+
) -> _CaseOutcome:
|
|
3328
|
+
transcript = "\n".join(
|
|
3329
|
+
f"{message['role']}: {message['content']}" for message in messages
|
|
3330
|
+
)
|
|
3331
|
+
if (
|
|
3332
|
+
stop_reason in {"target_disconnected", "room_disconnected"}
|
|
3333
|
+
and _has_role_alternation(messages)
|
|
3334
|
+
and _has_natural_terminal_exchange(messages)
|
|
3335
|
+
):
|
|
3336
|
+
# A provider target can deliberately end a short call before the generated
|
|
3337
|
+
# minimum-turn budget (for example, by accepting a caller's request to hang
|
|
3338
|
+
# up). The transport still completed successfully. Keep call lifecycle
|
|
3339
|
+
# separate from business-goal correctness: the scenario checks/evals decide
|
|
3340
|
+
# whether ending early was acceptable instead of reporting a false
|
|
3341
|
+
# connectivity failure.
|
|
3342
|
+
return _CaseOutcome(
|
|
3343
|
+
status=TestCaseStatus.COMPLETED,
|
|
3344
|
+
transcript=transcript,
|
|
3345
|
+
messages=messages,
|
|
3346
|
+
metadata={
|
|
3347
|
+
"stop_reason": stop_reason,
|
|
3348
|
+
"short_terminal_exchange": True,
|
|
3349
|
+
},
|
|
3350
|
+
)
|
|
3351
|
+
if (
|
|
3352
|
+
stop_reason == "conversation_silence_timeout"
|
|
3353
|
+
and len(messages) >= min_turn_messages
|
|
3354
|
+
and _has_role_alternation(messages)
|
|
3355
|
+
and _has_natural_terminal_exchange(messages)
|
|
3356
|
+
):
|
|
3357
|
+
# Agent-first calls use a short silence watchdog because the tested
|
|
3358
|
+
# agent owns the opening turn. A simulator can occasionally omit its
|
|
3359
|
+
# endCall tool even after both sides have clearly closed the call. Do
|
|
3360
|
+
# not turn a fully recorded farewell/transfer into an infrastructure
|
|
3361
|
+
# failure merely because the now-idle room remained open. Evaluation
|
|
3362
|
+
# still decides whether the agent actually completed the requested
|
|
3363
|
+
# business action.
|
|
3364
|
+
return _CaseOutcome(
|
|
3365
|
+
status=TestCaseStatus.COMPLETED,
|
|
3366
|
+
transcript=transcript,
|
|
3367
|
+
messages=messages,
|
|
3368
|
+
metadata={
|
|
3369
|
+
"stop_reason": stop_reason,
|
|
3370
|
+
"terminal_exchange_recovered": True,
|
|
3371
|
+
},
|
|
3372
|
+
)
|
|
3373
|
+
if stop_reason == "timeout":
|
|
3374
|
+
return _failure_outcome(
|
|
3375
|
+
TestCaseStatus.TIMED_OUT,
|
|
3376
|
+
FailureStage.RUNNING,
|
|
3377
|
+
"conversation_timeout",
|
|
3378
|
+
"Conversation exceeded its deadline",
|
|
3379
|
+
transcript=transcript,
|
|
3380
|
+
messages=messages,
|
|
3381
|
+
retryable=True,
|
|
3382
|
+
)
|
|
3383
|
+
if stop_reason == "conversation_silence_timeout" and _caller_never_spoke(messages):
|
|
3384
|
+
# The target sat in real silence because nothing was ever spoken at it. Retrying cannot
|
|
3385
|
+
# put a voice back on the line, so fail fast and name the synthesis rather than spending
|
|
3386
|
+
# the attempt budget reporting the agent as stalled.
|
|
3387
|
+
return _failure_outcome(
|
|
3388
|
+
TestCaseStatus.FAILED,
|
|
3389
|
+
FailureStage.RUNNING,
|
|
3390
|
+
"simulator_tts_silent",
|
|
3391
|
+
"Simulated caller produced transcript text but no audio",
|
|
3392
|
+
transcript=transcript,
|
|
3393
|
+
messages=messages,
|
|
3394
|
+
retryable=False,
|
|
3395
|
+
details={"stop_reason": stop_reason, "turn_count": str(len(messages))},
|
|
3396
|
+
)
|
|
3397
|
+
stalled = {
|
|
3398
|
+
"conversation_silence_timeout",
|
|
3399
|
+
"conversation_stalled",
|
|
3400
|
+
"session_closed",
|
|
3401
|
+
"no_conversation",
|
|
3402
|
+
"monitor_failed",
|
|
3403
|
+
}
|
|
3404
|
+
if _answered_by_voicemail():
|
|
3405
|
+
# Silence after a mailbox greeting is the call's natural end, not a stall.
|
|
3406
|
+
stalled.discard("conversation_silence_timeout")
|
|
3407
|
+
if stop_reason in stalled:
|
|
3408
|
+
code = stop_reason
|
|
3409
|
+
message = {
|
|
3410
|
+
"conversation_silence_timeout": (
|
|
3411
|
+
"Agent-first conversation stalled after it began"
|
|
3412
|
+
),
|
|
3413
|
+
"conversation_stalled": (
|
|
3414
|
+
"Conversation produced no new speech for the stall deadline"
|
|
3415
|
+
),
|
|
3416
|
+
"session_closed": (
|
|
3417
|
+
"Conversation session closed before a natural end condition"
|
|
3418
|
+
),
|
|
3419
|
+
"no_conversation": (
|
|
3420
|
+
"No conversation turns were committed before the inactivity deadline"
|
|
3421
|
+
),
|
|
3422
|
+
"monitor_failed": (
|
|
3423
|
+
"Conversation end monitoring failed before a natural end condition"
|
|
3424
|
+
),
|
|
3425
|
+
}[stop_reason]
|
|
3426
|
+
return _failure_outcome(
|
|
3427
|
+
TestCaseStatus.FAILED,
|
|
3428
|
+
FailureStage.RUNNING,
|
|
3429
|
+
code,
|
|
3430
|
+
message,
|
|
3431
|
+
transcript=transcript,
|
|
3432
|
+
messages=messages,
|
|
3433
|
+
retryable=True,
|
|
3434
|
+
)
|
|
3435
|
+
floor, alternation_required = _turn_requirements(min_turn_messages)
|
|
3436
|
+
if len(messages) < floor or (
|
|
3437
|
+
alternation_required and not _has_role_alternation(messages)
|
|
3438
|
+
):
|
|
3439
|
+
code = (
|
|
3440
|
+
stop_reason
|
|
3441
|
+
if stop_reason in {"target_disconnected", "room_disconnected"}
|
|
3442
|
+
else "insufficient_conversation"
|
|
3443
|
+
)
|
|
3444
|
+
return _failure_outcome(
|
|
3445
|
+
TestCaseStatus.FAILED,
|
|
3446
|
+
FailureStage.RUNNING,
|
|
3447
|
+
code,
|
|
3448
|
+
"Conversation ended before the required alternating turns completed",
|
|
3449
|
+
transcript=transcript,
|
|
3450
|
+
messages=messages,
|
|
3451
|
+
retryable=stop_reason
|
|
3452
|
+
in {"target_disconnected", "room_disconnected", "session_closed"},
|
|
3453
|
+
details={
|
|
3454
|
+
"stop_reason": stop_reason,
|
|
3455
|
+
"turn_count": str(len(messages)),
|
|
3456
|
+
"minimum_turn_count": str(floor),
|
|
3457
|
+
},
|
|
3458
|
+
)
|
|
3459
|
+
return _CaseOutcome(
|
|
3460
|
+
status=TestCaseStatus.COMPLETED,
|
|
3461
|
+
transcript=transcript,
|
|
3462
|
+
messages=messages,
|
|
3463
|
+
metadata={"stop_reason": stop_reason},
|
|
3464
|
+
)
|
|
3465
|
+
|
|
3466
|
+
|
|
3467
|
+
def _has_natural_terminal_exchange(messages: list[dict[str, str]]) -> bool:
|
|
3468
|
+
"""Recognize only explicit terminal language near the end of a call.
|
|
3469
|
+
|
|
3470
|
+
This deliberately avoids broad sentiment or short-answer heuristics. A
|
|
3471
|
+
normal unanswered question must remain a silence failure. The two safe
|
|
3472
|
+
cases are an explicit farewell, or a transfer handoff followed by the
|
|
3473
|
+
caller's acknowledgement.
|
|
3474
|
+
"""
|
|
3475
|
+
tail = [
|
|
3476
|
+
(
|
|
3477
|
+
str(message.get("role") or "").lower(),
|
|
3478
|
+
str(message.get("content") or "").strip().lower(),
|
|
3479
|
+
)
|
|
3480
|
+
for message in messages[-4:]
|
|
3481
|
+
if str(message.get("content") or "").strip()
|
|
3482
|
+
]
|
|
3483
|
+
if not tail:
|
|
3484
|
+
return False
|
|
3485
|
+
farewell_markers = (
|
|
3486
|
+
"goodbye",
|
|
3487
|
+
"bye",
|
|
3488
|
+
"take care",
|
|
3489
|
+
"have a great day",
|
|
3490
|
+
"have a good day",
|
|
3491
|
+
"have a nice day",
|
|
3492
|
+
)
|
|
3493
|
+
if any(marker in text for _role, text in tail for marker in farewell_markers):
|
|
3494
|
+
return True
|
|
3495
|
+
|
|
3496
|
+
for index, (role, text) in enumerate(tail[:-1]):
|
|
3497
|
+
if role != "assistant" or "transfer" not in text:
|
|
3498
|
+
continue
|
|
3499
|
+
if not any(marker in text for marker in ("now", "connect", "please wait")):
|
|
3500
|
+
continue
|
|
3501
|
+
next_role, acknowledgement = tail[index + 1]
|
|
3502
|
+
if next_role == "user" and acknowledgement.rstrip(".! ") in {
|
|
3503
|
+
"ok",
|
|
3504
|
+
"okay",
|
|
3505
|
+
"alright",
|
|
3506
|
+
"please do",
|
|
3507
|
+
"thank you",
|
|
3508
|
+
"thanks",
|
|
3509
|
+
}:
|
|
3510
|
+
return True
|
|
3511
|
+
return False
|
|
3512
|
+
|
|
3513
|
+
|
|
3514
|
+
def _failure_outcome(
|
|
3515
|
+
status: TestCaseStatus,
|
|
3516
|
+
stage: FailureStage,
|
|
3517
|
+
code: str,
|
|
3518
|
+
message: str,
|
|
3519
|
+
*,
|
|
3520
|
+
transcript: str = "",
|
|
3521
|
+
messages: list[dict[str, str]] | None = None,
|
|
3522
|
+
retryable: bool = False,
|
|
3523
|
+
details: dict[str, str] | None = None,
|
|
3524
|
+
) -> _CaseOutcome:
|
|
3525
|
+
return _CaseOutcome(
|
|
3526
|
+
status=status,
|
|
3527
|
+
transcript=transcript,
|
|
3528
|
+
messages=messages or [],
|
|
3529
|
+
failure=SimulationFailure(
|
|
3530
|
+
stage=stage,
|
|
3531
|
+
code=code,
|
|
3532
|
+
message=message,
|
|
3533
|
+
retryable=retryable,
|
|
3534
|
+
provider="livekit",
|
|
3535
|
+
details=details or {},
|
|
3536
|
+
),
|
|
3537
|
+
)
|
|
3538
|
+
|
|
3539
|
+
|
|
3540
|
+
def _record_simulator_setup(
|
|
3541
|
+
case_directory: Path,
|
|
3542
|
+
*,
|
|
3543
|
+
persona: Persona,
|
|
3544
|
+
instructions: str,
|
|
3545
|
+
llm_config: Any,
|
|
3546
|
+
stt_config: Any,
|
|
3547
|
+
tts_config: Any,
|
|
3548
|
+
turn_handling: Any = None,
|
|
3549
|
+
extra: dict[str, Any] | None = None,
|
|
3550
|
+
) -> None:
|
|
3551
|
+
"""Write the exact prompt and voice settings this call is about to use.
|
|
3552
|
+
|
|
3553
|
+
Reconstructing either one afterwards from a transcript is guesswork, and the simulator's
|
|
3554
|
+
prompt is what decides how the caller behaves. Written before the call connects so it
|
|
3555
|
+
survives a run that dies mid-conversation.
|
|
3556
|
+
"""
|
|
3557
|
+
|
|
3558
|
+
def settings(config: Any) -> Any:
|
|
3559
|
+
if config is None:
|
|
3560
|
+
return None
|
|
3561
|
+
for method in ("model_dump", "dict"):
|
|
3562
|
+
dump = getattr(config, method, None)
|
|
3563
|
+
if callable(dump):
|
|
3564
|
+
try:
|
|
3565
|
+
return dump()
|
|
3566
|
+
except Exception: # noqa: BLE001 - never fail a call over logging
|
|
3567
|
+
pass
|
|
3568
|
+
return str(config)
|
|
3569
|
+
|
|
3570
|
+
try:
|
|
3571
|
+
case_directory.mkdir(parents=True, exist_ok=True)
|
|
3572
|
+
(case_directory / "simulator-prompt.txt").write_text(
|
|
3573
|
+
instructions or "", encoding="utf-8"
|
|
3574
|
+
)
|
|
3575
|
+
payload = {
|
|
3576
|
+
"persona": settings(persona),
|
|
3577
|
+
"simulator_system_prompt": instructions or "",
|
|
3578
|
+
"llm": settings(llm_config),
|
|
3579
|
+
"stt": settings(stt_config),
|
|
3580
|
+
"tts": settings(tts_config),
|
|
3581
|
+
"turn_handling": settings(turn_handling),
|
|
3582
|
+
}
|
|
3583
|
+
payload.update(extra or {})
|
|
3584
|
+
(case_directory / "simulator-setup.json").write_text(
|
|
3585
|
+
json.dumps(payload, indent=2, ensure_ascii=False, default=str),
|
|
3586
|
+
encoding="utf-8",
|
|
3587
|
+
)
|
|
3588
|
+
except Exception as error: # noqa: BLE001 - logging must never break a run
|
|
3589
|
+
logger.warning("could not record simulator setup: %s", error)
|
|
3590
|
+
|
|
3591
|
+
|
|
3592
|
+
def _attach_recordings(
|
|
3593
|
+
outcome: _CaseOutcome,
|
|
3594
|
+
recorder: RoomRecorder,
|
|
3595
|
+
*,
|
|
3596
|
+
simulator_identity: str,
|
|
3597
|
+
target_identity: str | None,
|
|
3598
|
+
target_track_sid: str | None,
|
|
3599
|
+
case_directory: Path,
|
|
3600
|
+
sample_rate: int,
|
|
3601
|
+
) -> None:
|
|
3602
|
+
simulator_paths = recorder.paths_for_participant(simulator_identity)
|
|
3603
|
+
target_paths = (
|
|
3604
|
+
recorder.paths_for_participant(
|
|
3605
|
+
target_identity,
|
|
3606
|
+
track_sid=target_track_sid,
|
|
3607
|
+
)
|
|
3608
|
+
if target_identity is not None
|
|
3609
|
+
else []
|
|
3610
|
+
)
|
|
3611
|
+
# ``audio_track_sid`` identifies the track used to establish target
|
|
3612
|
+
# readiness, but it is not necessarily the track that remains published for
|
|
3613
|
+
# the conversation. Agents using ``BackgroundAudioPlayer`` publish more
|
|
3614
|
+
# than one audio track and can replace the initially selected publication.
|
|
3615
|
+
# Recording is evidence of the whole participant, so fall back to all of the
|
|
3616
|
+
# target participant's tracks instead of silently producing no artifact.
|
|
3617
|
+
if target_identity is not None and not target_paths:
|
|
3618
|
+
target_paths = recorder.paths_for_participant(target_identity)
|
|
3619
|
+
|
|
3620
|
+
# The simulator normally publishes with ``simulator_identity``. Retain a
|
|
3621
|
+
# conservative fallback for SDKs that expose the local publication under a
|
|
3622
|
+
# different participant identity: only use it when there is exactly one
|
|
3623
|
+
# non-target publishing participant, so another caller can never be folded
|
|
3624
|
+
# into the customer channel accidentally.
|
|
3625
|
+
if not simulator_paths:
|
|
3626
|
+
non_target_identities = {
|
|
3627
|
+
record.participant_identity
|
|
3628
|
+
for record in recorder.records
|
|
3629
|
+
if record.participant_identity != target_identity
|
|
3630
|
+
}
|
|
3631
|
+
if len(non_target_identities) == 1:
|
|
3632
|
+
simulator_paths = recorder.paths_for_participant(
|
|
3633
|
+
next(iter(non_target_identities))
|
|
3634
|
+
)
|
|
3635
|
+
audio_directory = case_directory / "audio"
|
|
3636
|
+
input_path = _collapse_recordings(
|
|
3637
|
+
simulator_paths,
|
|
3638
|
+
audio_directory / "simulator.wav",
|
|
3639
|
+
sample_rate=sample_rate,
|
|
3640
|
+
)
|
|
3641
|
+
output_path = _collapse_recordings(
|
|
3642
|
+
target_paths,
|
|
3643
|
+
audio_directory / "target.wav",
|
|
3644
|
+
sample_rate=sample_rate,
|
|
3645
|
+
)
|
|
3646
|
+
combined_path = mix_recordings(
|
|
3647
|
+
[path for path in (input_path, output_path) if path is not None],
|
|
3648
|
+
audio_directory / "combined.wav",
|
|
3649
|
+
sample_rate=sample_rate,
|
|
3650
|
+
)
|
|
3651
|
+
stereo_path = mix_recordings_stereo(
|
|
3652
|
+
[path for path in (input_path,) if path is not None],
|
|
3653
|
+
[path for path in (output_path,) if path is not None],
|
|
3654
|
+
audio_directory / "stereo.wav",
|
|
3655
|
+
sample_rate=sample_rate,
|
|
3656
|
+
)
|
|
3657
|
+
outcome.audio_input_path = str(input_path) if input_path is not None else None
|
|
3658
|
+
outcome.audio_output_path = str(output_path) if output_path is not None else None
|
|
3659
|
+
outcome.audio_combined_path = (
|
|
3660
|
+
str(combined_path) if combined_path is not None else None
|
|
3661
|
+
)
|
|
3662
|
+
outcome.audio_stereo_path = str(stereo_path) if stereo_path is not None else None
|
|
3663
|
+
outcome.metadata["recording_tracks"] = [
|
|
3664
|
+
{
|
|
3665
|
+
"participant_identity": record.participant_identity,
|
|
3666
|
+
"participant_sid": record.participant_sid,
|
|
3667
|
+
"track_sid": record.track_sid,
|
|
3668
|
+
"path": str(record.path),
|
|
3669
|
+
"start_offset_frames": record.start_offset_frames,
|
|
3670
|
+
}
|
|
3671
|
+
for record in recorder.records
|
|
3672
|
+
]
|
|
3673
|
+
outcome.metadata["recording_diagnostics"] = {
|
|
3674
|
+
"simulator_identity": simulator_identity,
|
|
3675
|
+
"target_identity": target_identity,
|
|
3676
|
+
"target_track_sid": target_track_sid,
|
|
3677
|
+
"simulator_track_count": len(simulator_paths),
|
|
3678
|
+
"target_track_count": len(target_paths),
|
|
3679
|
+
"recorder_error_types": [type(error).__name__ for error in recorder.errors],
|
|
3680
|
+
}
|
|
3681
|
+
if combined_path is None:
|
|
3682
|
+
logger.warning(
|
|
3683
|
+
"LiveKit call completed without captured audio tracks",
|
|
3684
|
+
extra={
|
|
3685
|
+
"simulator_identity": simulator_identity,
|
|
3686
|
+
"target_identity": target_identity,
|
|
3687
|
+
"target_track_sid": target_track_sid,
|
|
3688
|
+
"recorded_track_count": len(recorder.records),
|
|
3689
|
+
"recorder_error_types": [
|
|
3690
|
+
type(error).__name__ for error in recorder.errors
|
|
3691
|
+
],
|
|
3692
|
+
},
|
|
3693
|
+
)
|
|
3694
|
+
speech_starts = [
|
|
3695
|
+
float(message["started_speaking_at"])
|
|
3696
|
+
for message in outcome.messages
|
|
3697
|
+
if isinstance(message.get("started_speaking_at"), (int, float))
|
|
3698
|
+
]
|
|
3699
|
+
if recorder.recording_started_at is not None and speech_starts:
|
|
3700
|
+
outcome.metadata["recording_offset_ms"] = max(
|
|
3701
|
+
0,
|
|
3702
|
+
round((min(speech_starts) - recorder.recording_started_at) * 1000),
|
|
3703
|
+
)
|
|
3704
|
+
|
|
3705
|
+
|
|
3706
|
+
def _collapse_recordings(
|
|
3707
|
+
paths: list[Path],
|
|
3708
|
+
destination: Path,
|
|
3709
|
+
*,
|
|
3710
|
+
sample_rate: int,
|
|
3711
|
+
) -> Path | None:
|
|
3712
|
+
if not paths:
|
|
3713
|
+
return None
|
|
3714
|
+
if len(paths) == 1:
|
|
3715
|
+
return paths[0]
|
|
3716
|
+
return mix_recordings(paths, destination, sample_rate=sample_rate)
|
|
3717
|
+
|
|
3718
|
+
|
|
3719
|
+
def _target_api_key(target: VoiceProviderTarget | None) -> str | None:
|
|
3720
|
+
if target is None:
|
|
3721
|
+
return None
|
|
3722
|
+
return os.environ.get(target.api_key_env) or None
|
|
3723
|
+
|
|
3724
|
+
|
|
3725
|
+
def _target_evidence_base_url(target: VoiceProviderTarget | None) -> str | None:
|
|
3726
|
+
if isinstance(target, VapiTargetConfig):
|
|
3727
|
+
return str(target.api_base_url).rstrip("/")
|
|
3728
|
+
if isinstance(target, RetellTargetConfig):
|
|
3729
|
+
parsed = urlsplit(str(target.api_url))
|
|
3730
|
+
return f"{parsed.scheme}://{parsed.netloc}"
|
|
3731
|
+
return None
|
|
3732
|
+
|
|
3733
|
+
|
|
3734
|
+
def _default_simulator_llm_config() -> LLMConfig:
|
|
3735
|
+
return LLMConfig(
|
|
3736
|
+
provider=os.environ.get("SIMULATOR_LLM_PROVIDER", "openai"),
|
|
3737
|
+
model=os.environ.get("SIMULATOR_LLM_MODEL", "gpt-4o-mini"),
|
|
3738
|
+
temperature=0.6,
|
|
3739
|
+
)
|
|
3740
|
+
|
|
3741
|
+
|
|
3742
|
+
def _resolve_livekit_runtime(
|
|
3743
|
+
agent_definition: AgentDefinition,
|
|
3744
|
+
runtime: LiveKitSimulatorRuntime | None,
|
|
3745
|
+
) -> LiveKitSimulatorRuntime:
|
|
3746
|
+
if runtime is not None:
|
|
3747
|
+
return runtime
|
|
3748
|
+
if agent_definition.url is None or not agent_definition.room_name:
|
|
3749
|
+
raise ValueError(
|
|
3750
|
+
"livekit_runtime_required: provide LiveKitSimulatorRuntime or legacy "
|
|
3751
|
+
"AgentDefinition url and room_name"
|
|
3752
|
+
)
|
|
3753
|
+
return LiveKitSimulatorRuntime(
|
|
3754
|
+
url=agent_definition.url,
|
|
3755
|
+
room_name=agent_definition.room_name,
|
|
3756
|
+
room_mode=agent_definition.room_mode,
|
|
3757
|
+
)
|
|
3758
|
+
|
|
3759
|
+
|
|
3760
|
+
def _resolve_room_name(
|
|
3761
|
+
runtime: LiveKitSimulatorRuntime,
|
|
3762
|
+
*,
|
|
3763
|
+
run_id: str,
|
|
3764
|
+
test_case_id: str,
|
|
3765
|
+
index: int,
|
|
3766
|
+
invocation_id: str,
|
|
3767
|
+
) -> str:
|
|
3768
|
+
rendered = runtime.room_name.format(
|
|
3769
|
+
run_id=run_id,
|
|
3770
|
+
test_case_id=test_case_id,
|
|
3771
|
+
index=index,
|
|
3772
|
+
invocation_id=invocation_id,
|
|
3773
|
+
)
|
|
3774
|
+
if runtime.room_mode == "external" or getattr(runtime, "room_name_verbatim", False):
|
|
3775
|
+
return rendered
|
|
3776
|
+
prefix = _SAFE_ROOM.sub("-", rendered).strip("-._") or "simulation"
|
|
3777
|
+
suffix_parts = []
|
|
3778
|
+
if invocation_id not in prefix:
|
|
3779
|
+
suffix_parts.append(invocation_id)
|
|
3780
|
+
if test_case_id not in prefix:
|
|
3781
|
+
suffix_parts.append(test_case_id[-12:])
|
|
3782
|
+
suffix = "-" + "-".join(suffix_parts) if suffix_parts else ""
|
|
3783
|
+
return f"{prefix[: 255 - len(suffix)]}{suffix}"
|
|
3784
|
+
|
|
3785
|
+
|
|
3786
|
+
def _has_room_template(room_name: str) -> bool:
|
|
3787
|
+
return any(
|
|
3788
|
+
marker in room_name for marker in ("{run_id}", "{test_case_id}", "{index}")
|
|
3789
|
+
)
|
|
3790
|
+
|
|
3791
|
+
|
|
3792
|
+
def _api_url(url: str) -> str:
|
|
3793
|
+
if url.startswith("wss://"):
|
|
3794
|
+
return "https://" + url.removeprefix("wss://")
|
|
3795
|
+
if url.startswith("ws://"):
|
|
3796
|
+
return "http://" + url.removeprefix("ws://")
|
|
3797
|
+
return url
|
|
3798
|
+
|
|
3799
|
+
|
|
3800
|
+
def _remove_room_listener(room: rtc.Room, event: str, listener) -> None:
|
|
3801
|
+
try:
|
|
3802
|
+
room.off(event, listener)
|
|
3803
|
+
except (AttributeError, ValueError):
|
|
3804
|
+
logger.debug("LiveKit listener was already removed", extra={"event": event})
|
|
3805
|
+
|
|
3806
|
+
|
|
3807
|
+
async def _close_agent_session(session: AgentSession, *, timeout: float) -> None:
|
|
3808
|
+
"""Close a session without abandoning teardown on the event loop.
|
|
3809
|
+
|
|
3810
|
+
A shielded, timed-out ``aclose`` used to keep running after the case had
|
|
3811
|
+
returned. Repeating that in a soak test accumulated SDK activities until
|
|
3812
|
+
the guest process failed. Graceful close gets a bounded opportunity; after
|
|
3813
|
+
that, cancel and reap it because room and process teardown are independent.
|
|
3814
|
+
"""
|
|
3815
|
+
close_session = getattr(session, "aclose", None)
|
|
3816
|
+
if close_session is None:
|
|
3817
|
+
session.shutdown(drain=False)
|
|
3818
|
+
return
|
|
3819
|
+
close_task = asyncio.create_task(close_session())
|
|
3820
|
+
try:
|
|
3821
|
+
await asyncio.wait_for(asyncio.shield(close_task), timeout=timeout)
|
|
3822
|
+
except asyncio.TimeoutError:
|
|
3823
|
+
close_task.cancel()
|
|
3824
|
+
try:
|
|
3825
|
+
await asyncio.wait_for(close_task, timeout=1.0)
|
|
3826
|
+
except (Exception, asyncio.CancelledError):
|
|
3827
|
+
if not close_task.done():
|
|
3828
|
+
close_task.add_done_callback(_consume_background_task_result)
|
|
3829
|
+
raise
|
|
3830
|
+
|
|
3831
|
+
|
|
3832
|
+
def _consume_background_task_result(task: asyncio.Task) -> None:
|
|
3833
|
+
try:
|
|
3834
|
+
task.result()
|
|
3835
|
+
except (Exception, asyncio.CancelledError):
|
|
3836
|
+
pass
|
|
3837
|
+
|
|
3838
|
+
|
|
3839
|
+
def _is_not_found(exc: Exception) -> bool:
|
|
3840
|
+
code = getattr(exc, "code", None)
|
|
3841
|
+
return str(getattr(code, "value", code)).lower() in {
|
|
3842
|
+
"not_found",
|
|
3843
|
+
"404",
|
|
3844
|
+
}
|
|
3845
|
+
|
|
3846
|
+
|
|
3847
|
+
def _record_cleanup_error(
|
|
3848
|
+
errors: list[str],
|
|
3849
|
+
exc: Exception,
|
|
3850
|
+
operation: str,
|
|
3851
|
+
run_id: str,
|
|
3852
|
+
test_case_id: str,
|
|
3853
|
+
) -> None:
|
|
3854
|
+
errors.append(f"{operation}:{type(exc).__name__}")
|
|
3855
|
+
logger.error(
|
|
3856
|
+
"LiveKit cleanup operation failed",
|
|
3857
|
+
exc_info=redacted_exc_info(exc),
|
|
3858
|
+
extra={
|
|
3859
|
+
"run_id": run_id,
|
|
3860
|
+
"test_case_id": test_case_id,
|
|
3861
|
+
"operation": operation,
|
|
3862
|
+
"exception_type": type(exc).__name__,
|
|
3863
|
+
},
|
|
3864
|
+
)
|
|
3865
|
+
|
|
3866
|
+
|
|
3867
|
+
_LIVEKIT_INBOUND_TRUNK_ENV = "LIVEKIT_INBOUND_TRUNK_ID"
|
|
3868
|
+
|
|
3869
|
+
|
|
3870
|
+
def _safe_provider_error_details(
|
|
3871
|
+
exc: Exception, *, operation: str
|
|
3872
|
+
) -> dict[str, object]:
|
|
3873
|
+
"""Extract sanitized error attributes for report failures.
|
|
3874
|
+
|
|
3875
|
+
Never returns the exception message; only structural fields that are
|
|
3876
|
+
known to be safe from LiveKit/Twirp exception classes.
|
|
3877
|
+
"""
|
|
3878
|
+
|
|
3879
|
+
code = getattr(exc, "code", None)
|
|
3880
|
+
if code is not None:
|
|
3881
|
+
code_value = getattr(code, "value", None)
|
|
3882
|
+
if code_value is None and not isinstance(code, (str, int)):
|
|
3883
|
+
code_value = str(code)
|
|
3884
|
+
else:
|
|
3885
|
+
code_value = code_value if code_value is not None else code
|
|
3886
|
+
else:
|
|
3887
|
+
code_value = None
|
|
3888
|
+
status = getattr(exc, "status", None) or getattr(exc, "status_code", None)
|
|
3889
|
+
details: dict[str, object] = {
|
|
3890
|
+
"operation": operation,
|
|
3891
|
+
"exception_type": type(exc).__name__,
|
|
3892
|
+
}
|
|
3893
|
+
if code_value is not None:
|
|
3894
|
+
details["provider_code"] = code_value
|
|
3895
|
+
if status is not None:
|
|
3896
|
+
try:
|
|
3897
|
+
details["http_status"] = int(status)
|
|
3898
|
+
except (TypeError, ValueError):
|
|
3899
|
+
details["http_status"] = str(status)
|
|
3900
|
+
metadata = getattr(exc, "metadata", None)
|
|
3901
|
+
if isinstance(metadata, dict):
|
|
3902
|
+
for key in ("sip_status_code", "sip_status", "sip-code"):
|
|
3903
|
+
value = metadata.get(key)
|
|
3904
|
+
if value is not None:
|
|
3905
|
+
details["sip_status_code"] = str(value)
|
|
3906
|
+
break
|
|
3907
|
+
return details
|
|
3908
|
+
|
|
3909
|
+
|
|
3910
|
+
async def _ensure_sip_inbound_dispatch(
|
|
3911
|
+
api_client: api.LiveKitAPI,
|
|
3912
|
+
*,
|
|
3913
|
+
transport: TelephonyTransport,
|
|
3914
|
+
room_name: str,
|
|
3915
|
+
) -> tuple[str, bool]:
|
|
3916
|
+
"""Return ``(sip_dispatch_rule_id, created_by_sdk)``.
|
|
3917
|
+
|
|
3918
|
+
When ``transport.dispatch_rule_name`` is supplied the SDK verifies
|
|
3919
|
+
the rule exists and reuses it. Otherwise the SDK provisions a
|
|
3920
|
+
per-run direct rule bound to ``LIVEKIT_INBOUND_TRUNK_ID`` that routes
|
|
3921
|
+
incoming calls into ``room_name`` — the same room the local
|
|
3922
|
+
simulator has already joined — and returns its id so the caller can
|
|
3923
|
+
tear it down.
|
|
3924
|
+
"""
|
|
3925
|
+
|
|
3926
|
+
existing = await api_client.sip.list_sip_dispatch_rule(ListSIPDispatchRuleRequest())
|
|
3927
|
+
if transport.dispatch_rule_name:
|
|
3928
|
+
for rule in existing.items:
|
|
3929
|
+
if rule.name != transport.dispatch_rule_name:
|
|
3930
|
+
continue
|
|
3931
|
+
direct = (
|
|
3932
|
+
getattr(rule.rule, "dispatch_rule_direct", None) if rule.rule else None
|
|
3933
|
+
)
|
|
3934
|
+
direct_room = getattr(direct, "room_name", "") if direct is not None else ""
|
|
3935
|
+
if not direct_room:
|
|
3936
|
+
raise RuntimeError(
|
|
3937
|
+
"sip_inbound_rule_mismatch: "
|
|
3938
|
+
f"{transport.dispatch_rule_name} is not a direct rule"
|
|
3939
|
+
)
|
|
3940
|
+
if direct_room != room_name:
|
|
3941
|
+
raise RuntimeError(
|
|
3942
|
+
"sip_inbound_rule_mismatch: "
|
|
3943
|
+
f"{transport.dispatch_rule_name} targets a different room"
|
|
3944
|
+
)
|
|
3945
|
+
return rule.sip_dispatch_rule_id, False
|
|
3946
|
+
raise RuntimeError(f"sip_inbound_rule_missing: {transport.dispatch_rule_name}")
|
|
3947
|
+
trunk_id = os.environ.get(_LIVEKIT_INBOUND_TRUNK_ENV)
|
|
3948
|
+
if not trunk_id:
|
|
3949
|
+
raise RuntimeError(
|
|
3950
|
+
f"sip_inbound_trunk_missing: set {_LIVEKIT_INBOUND_TRUNK_ENV}"
|
|
3951
|
+
)
|
|
3952
|
+
for rule in existing.items:
|
|
3953
|
+
if trunk_id and trunk_id in rule.trunk_ids:
|
|
3954
|
+
raise RuntimeError(
|
|
3955
|
+
"sip_inbound_route_conflict: existing dispatch rule "
|
|
3956
|
+
f"{rule.sip_dispatch_rule_id} already covers this trunk"
|
|
3957
|
+
)
|
|
3958
|
+
rule_name = f"sim-inbound-{room_name[-24:]}"
|
|
3959
|
+
resp = await api_client.sip.create_sip_dispatch_rule(
|
|
3960
|
+
CreateSIPDispatchRuleRequest(
|
|
3961
|
+
rule=SIPDispatchRule(
|
|
3962
|
+
dispatch_rule_direct=SIPDispatchRuleDirect(
|
|
3963
|
+
room_name=room_name,
|
|
3964
|
+
),
|
|
3965
|
+
),
|
|
3966
|
+
trunk_ids=[trunk_id],
|
|
3967
|
+
hide_phone_number=False,
|
|
3968
|
+
name=rule_name,
|
|
3969
|
+
)
|
|
3970
|
+
)
|
|
3971
|
+
return resp.sip_dispatch_rule_id, True
|
|
3972
|
+
|
|
3973
|
+
|
|
3974
|
+
async def _ensure_room_absent(
|
|
3975
|
+
api_client: api.LiveKitAPI, room_name: str, *, poll_interval: float = 0.5
|
|
3976
|
+
) -> int:
|
|
3977
|
+
"""Make ``room_name`` absent before the caller (re-)creates it.
|
|
3978
|
+
|
|
3979
|
+
The pool's dispatch rule stays live for the whole run, so an inbound call
|
|
3980
|
+
can re-create this room between our own delete and our next poll — hence
|
|
3981
|
+
the re-delete inside the loop rather than a single delete-then-poll. No
|
|
3982
|
+
internal deadline: the caller bounds this with ``asyncio.wait_for``.
|
|
3983
|
+
"""
|
|
3984
|
+
|
|
3985
|
+
try:
|
|
3986
|
+
await api_client.room.delete_room(api.DeleteRoomRequest(room=room_name))
|
|
3987
|
+
except Exception as exc: # noqa: BLE001
|
|
3988
|
+
if not _is_not_found(exc):
|
|
3989
|
+
raise
|
|
3990
|
+
polls = 0
|
|
3991
|
+
while True:
|
|
3992
|
+
resp = await api_client.room.list_rooms(api.ListRoomsRequest(names=[room_name]))
|
|
3993
|
+
polls += 1
|
|
3994
|
+
if not resp.rooms:
|
|
3995
|
+
return polls
|
|
3996
|
+
try:
|
|
3997
|
+
await api_client.room.delete_room(api.DeleteRoomRequest(room=room_name))
|
|
3998
|
+
except Exception as exc: # noqa: BLE001
|
|
3999
|
+
if not _is_not_found(exc):
|
|
4000
|
+
raise
|
|
4001
|
+
await asyncio.sleep(poll_interval if polls < 4 else 5.0)
|
|
4002
|
+
|
|
4003
|
+
|
|
4004
|
+
def _unexpected_participants(
|
|
4005
|
+
room, *, simulator_identity: str, recorder_identity: str
|
|
4006
|
+
) -> set[str]:
|
|
4007
|
+
return {str(p.identity) for p in room.remote_participants.values()} - {
|
|
4008
|
+
simulator_identity,
|
|
4009
|
+
recorder_identity,
|
|
4010
|
+
}
|
|
4011
|
+
|
|
4012
|
+
|
|
4013
|
+
# LiveKit: the other party's number on a SIP participant (the caller, for an
|
|
4014
|
+
# inbound call). sip.trunkPhoneNumber is OUR number — never use it.
|
|
4015
|
+
_SIP_REMOTE_NUMBER_ATTRIBUTE = "sip.phoneNumber"
|
|
4016
|
+
|
|
4017
|
+
|
|
4018
|
+
def _number_digits(value) -> str:
|
|
4019
|
+
text = str(value or "")
|
|
4020
|
+
if text.startswith("sip:"):
|
|
4021
|
+
text = text[len("sip:") :]
|
|
4022
|
+
text = text.split("@", 1)[0]
|
|
4023
|
+
text = text.split(";", 1)[0]
|
|
4024
|
+
return re.sub(r"\D", "", text)
|
|
4025
|
+
|
|
4026
|
+
|
|
4027
|
+
def _caller_matches(attributes, expected: str | None) -> bool | None:
|
|
4028
|
+
if not expected:
|
|
4029
|
+
return None
|
|
4030
|
+
observed = attributes.get(_SIP_REMOTE_NUMBER_ATTRIBUTE) if attributes else None
|
|
4031
|
+
if not observed:
|
|
4032
|
+
return None
|
|
4033
|
+
expected_digits = _number_digits(expected)
|
|
4034
|
+
observed_digits = _number_digits(observed)
|
|
4035
|
+
if len(expected_digits) < 7 or len(observed_digits) < 7:
|
|
4036
|
+
return None
|
|
4037
|
+
longer, shorter = (
|
|
4038
|
+
(expected_digits, observed_digits)
|
|
4039
|
+
if len(expected_digits) >= len(observed_digits)
|
|
4040
|
+
else (observed_digits, expected_digits)
|
|
4041
|
+
)
|
|
4042
|
+
return longer.endswith(shorter)
|
|
4043
|
+
|
|
4044
|
+
|
|
4045
|
+
class _LeasedRoomCallerMismatch(Exception):
|
|
4046
|
+
"""Module-private: raised by the leased-room caller check (step 4a), caught
|
|
4047
|
+
by the case's top-level try so the post-try recordings/evidence/metadata
|
|
4048
|
+
block still runs for a call that was billed and answered."""
|
|
4049
|
+
|
|
4050
|
+
|
|
4051
|
+
async def _delete_sip_dispatch_rule(api_client: api.LiveKitAPI, rule_id: str) -> None:
|
|
4052
|
+
await api_client.sip.delete_sip_dispatch_rule(
|
|
4053
|
+
DeleteSIPDispatchRuleRequest(sip_dispatch_rule_id=rule_id)
|
|
4054
|
+
)
|
|
4055
|
+
|
|
4056
|
+
|
|
4057
|
+
async def _collect_provider_evidence(
|
|
4058
|
+
*,
|
|
4059
|
+
config: ProviderEvidenceConfig,
|
|
4060
|
+
transport: TelephonyTransport,
|
|
4061
|
+
run_id: str,
|
|
4062
|
+
test_case_id: str,
|
|
4063
|
+
case_directory: Path,
|
|
4064
|
+
started_at: datetime,
|
|
4065
|
+
target: _TargetParticipant | None,
|
|
4066
|
+
provider_call_id_hint: str | None = None,
|
|
4067
|
+
provider_api_key: str | None = None,
|
|
4068
|
+
provider_api_base_url: str | None = None,
|
|
4069
|
+
termination_source: str | None = None,
|
|
4070
|
+
) -> tuple[EvidenceSourceSummary | None, list[ArtifactManifestEntry]]:
|
|
4071
|
+
call_id_hint = provider_call_id_hint
|
|
4072
|
+
caller_phone = transport.sip_number if transport.kind == "sip_outbound" else None
|
|
4073
|
+
callee_phone = transport.sip_call_to if transport.kind == "sip_outbound" else None
|
|
4074
|
+
if target is not None:
|
|
4075
|
+
if call_id_hint is None and config.participant_attribute:
|
|
4076
|
+
call_id_hint = target.attributes.get(config.participant_attribute)
|
|
4077
|
+
caller_phone = caller_phone or (
|
|
4078
|
+
target.attributes.get("sip.from")
|
|
4079
|
+
or target.attributes.get("sip.fromUser")
|
|
4080
|
+
or target.attributes.get("sip.callerNumber")
|
|
4081
|
+
)
|
|
4082
|
+
callee_phone = callee_phone or (
|
|
4083
|
+
target.attributes.get("sip.to")
|
|
4084
|
+
or target.attributes.get("sip.toUser")
|
|
4085
|
+
or target.attributes.get("sip.calledNumber")
|
|
4086
|
+
)
|
|
4087
|
+
context = EvidenceContext(
|
|
4088
|
+
run_id=run_id,
|
|
4089
|
+
test_case_id=test_case_id,
|
|
4090
|
+
case_directory=case_directory,
|
|
4091
|
+
started_at=started_at,
|
|
4092
|
+
call_id_hint=call_id_hint,
|
|
4093
|
+
caller_phone=caller_phone,
|
|
4094
|
+
callee_phone=callee_phone,
|
|
4095
|
+
termination_source=termination_source,
|
|
4096
|
+
)
|
|
4097
|
+
try:
|
|
4098
|
+
if config.provider == "vapi":
|
|
4099
|
+
adapter = VapiEvidenceSource(
|
|
4100
|
+
config,
|
|
4101
|
+
api_key=provider_api_key,
|
|
4102
|
+
api_base_url=provider_api_base_url,
|
|
4103
|
+
)
|
|
4104
|
+
elif config.provider == "retell":
|
|
4105
|
+
adapter = RetellEvidenceSource(
|
|
4106
|
+
config,
|
|
4107
|
+
api_key=provider_api_key,
|
|
4108
|
+
api_base_url=provider_api_base_url,
|
|
4109
|
+
)
|
|
4110
|
+
else:
|
|
4111
|
+
raise ProviderConfigError(
|
|
4112
|
+
f"unsupported_provider_evidence: {config.provider}"
|
|
4113
|
+
)
|
|
4114
|
+
except ProviderConfigError as exc:
|
|
4115
|
+
summary = EvidenceSourceSummary(
|
|
4116
|
+
source_id=f"{config.provider}:unconfigured",
|
|
4117
|
+
adapter=config.provider,
|
|
4118
|
+
evidence_class=_EVIDENCE_PROVIDER_REPORTED,
|
|
4119
|
+
available=False,
|
|
4120
|
+
redactions=["auth", "phone_e164"],
|
|
4121
|
+
metadata={"provider": config.provider, "reason": str(exc)},
|
|
4122
|
+
)
|
|
4123
|
+
return summary, []
|
|
4124
|
+
try:
|
|
4125
|
+
await adapter.connect(context)
|
|
4126
|
+
result: ProviderFetchResult = await adapter.fetch_final()
|
|
4127
|
+
except Exception as exc: # noqa: BLE001 — provider failures are first-class evidence
|
|
4128
|
+
logger.warning(
|
|
4129
|
+
"Provider evidence adapter failed",
|
|
4130
|
+
exc_info=redacted_exc_info(exc),
|
|
4131
|
+
extra={
|
|
4132
|
+
"provider": config.provider,
|
|
4133
|
+
"run_id": run_id,
|
|
4134
|
+
"test_case_id": test_case_id,
|
|
4135
|
+
},
|
|
4136
|
+
)
|
|
4137
|
+
summary = EvidenceSourceSummary(
|
|
4138
|
+
source_id=f"{config.provider}:error",
|
|
4139
|
+
adapter=config.provider,
|
|
4140
|
+
evidence_class=_EVIDENCE_PROVIDER_REPORTED,
|
|
4141
|
+
available=False,
|
|
4142
|
+
redactions=["auth", "phone_e164"],
|
|
4143
|
+
metadata={
|
|
4144
|
+
"provider": config.provider,
|
|
4145
|
+
"reason": "adapter_exception",
|
|
4146
|
+
"exception_type": type(exc).__name__,
|
|
4147
|
+
},
|
|
4148
|
+
)
|
|
4149
|
+
return summary, []
|
|
4150
|
+
finally:
|
|
4151
|
+
try:
|
|
4152
|
+
await adapter.close()
|
|
4153
|
+
except Exception as exc: # noqa: BLE001
|
|
4154
|
+
logger.debug(
|
|
4155
|
+
"Provider evidence adapter close failed",
|
|
4156
|
+
extra={
|
|
4157
|
+
"provider": config.provider,
|
|
4158
|
+
"exception_type": type(exc).__name__,
|
|
4159
|
+
},
|
|
4160
|
+
)
|
|
4161
|
+
return result.summary, result.artifacts
|
|
4162
|
+
|
|
4163
|
+
|
|
4164
|
+
# Import lazily to avoid a module-import cycle with ProviderConfigError above.
|
|
4165
|
+
from fi.simulate.evidence.base import EvidenceClass as _EvidenceClass # noqa: E402
|
|
4166
|
+
|
|
4167
|
+
_EVIDENCE_PROVIDER_REPORTED = _EvidenceClass.PROVIDER_REPORTED
|