agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,4167 @@
1
+ from __future__ import annotations
2
+
3
+ import array
4
+ import asyncio
5
+ import json
6
+ import math
7
+ import logging
8
+ import os
9
+ import re
10
+ import threading
11
+ import time
12
+ from dataclasses import dataclass, field
13
+ from urllib.parse import urlsplit
14
+ from pathlib import Path
15
+ from collections.abc import Awaitable, Callable
16
+ from typing import Any, AsyncIterable
17
+ from uuid import uuid4
18
+
19
+ try:
20
+ from livekit import api, rtc
21
+ from livekit.agents import (
22
+ Agent,
23
+ AgentSession,
24
+ AudioConfig,
25
+ BackgroundAudioPlayer,
26
+ RunContext,
27
+ function_tool,
28
+ metrics,
29
+ )
30
+ from livekit.agents.utils.audio import audio_frames_from_file
31
+ from livekit.agents.voice.background_audio import BuiltinAudioClip
32
+ from livekit.agents.types import (
33
+ ATTRIBUTE_TRANSCRIPTION_TRACK_ID,
34
+ TOPIC_TRANSCRIPTION,
35
+ )
36
+ from livekit.agents.voice import ModelSettings
37
+ from livekit.agents.voice.io import TimedString
38
+ from livekit.agents.voice.room_io import AudioInputOptions, RoomOptions
39
+ from livekit.api import AccessToken, VideoGrants
40
+ from livekit.plugins import silero
41
+ from livekit.protocol.sip import (
42
+ CreateSIPDispatchRuleRequest,
43
+ DeleteSIPDispatchRuleRequest,
44
+ ListSIPDispatchRuleRequest,
45
+ SIPDispatchRule,
46
+ SIPDispatchRuleDirect,
47
+ )
48
+ except ImportError as exc:
49
+ raise ImportError(
50
+ "LiveKit mode requires the 'livekit' optional dependency"
51
+ ) from exc
52
+
53
+ from datetime import datetime, timezone
54
+
55
+ from fi.simulate._logging import redacted_exc_info
56
+ from fi.simulate.agent.definition import (
57
+ AgentDefinition,
58
+ LiveKitSimulatorRuntime,
59
+ LLMConfig,
60
+ ProviderEvidenceConfig,
61
+ RetellTargetConfig,
62
+ SimulatorAgentDefinition,
63
+ STTConfig,
64
+ TelephonyTransport,
65
+ TTSConfig,
66
+ VapiTargetConfig,
67
+ VoiceProviderTarget,
68
+ )
69
+ from fi.simulate.artifacts.manifest import ArtifactManifestEntry
70
+ from fi.simulate.evidence.base import EvidenceSourceSummary
71
+ from fi.simulate.evidence.providers import (
72
+ EvidenceContext,
73
+ ProviderConfigError,
74
+ ProviderFetchResult,
75
+ RetellEvidenceSource,
76
+ VapiEvidenceSource,
77
+ )
78
+ from fi.simulate.endpoints.originators import (
79
+ CallOriginator,
80
+ build_call_originator,
81
+ finalize_originator,
82
+ )
83
+ from fi.simulate.simulation.bridge import LiveKitAudioBridge
84
+ from fi.simulate.simulation.bridge.audio import PCMResampler
85
+ from fi.simulate.simulation.livekit_models import LiveKitModels, build_livekit_models
86
+ from fi.simulate.recording.room_recorder import (
87
+ RoomRecorder,
88
+ mix_recordings,
89
+ mix_recordings_stereo,
90
+ )
91
+ from fi.simulate.runtime import (
92
+ FailureStage,
93
+ SimulationFailure,
94
+ TestCaseStatus,
95
+ derive_test_case_id,
96
+ new_run_id,
97
+ )
98
+ from fi.simulate.simulation.engines.base import BaseEngine
99
+ from fi.simulate.simulation.generator import ScenarioGenerator
100
+ from fi.simulate.simulation.models import Persona, Scenario, TestCaseResult, TestReport
101
+ from fi.simulate.simulation.voice_prompt import CallType, build_voice_simulator_prompt
102
+
103
+ logger = logging.getLogger(__name__)
104
+ _SAFE_ROOM = re.compile(r"[^A-Za-z0-9_.-]+")
105
+ # On conversation end, wait up to this long for the party still finishing its
106
+ # own turn to commit it (a LiveKit turn lands in history only after its TTS
107
+ # finishes playing), then delete the room so neither side keeps talking into a
108
+ # call the other has already left.
109
+ _FINAL_TURN_COMMIT_WAIT_SECONDS = 30.0
110
+ # The hosted platform inflates ``cleanup_timeout`` to carry the whole run
111
+ # budget (observed 1470s); as a per-step cleanup bound it must stay capped.
112
+ _MAX_CLEANUP_TIMEOUT_SECONDS = 60.0
113
+ # A dead LiveKit signal connection can leave any one SDK cleanup await pending
114
+ # indefinitely. The case-level deadline is still the outer bound, but no
115
+ # single best-effort operation may consume it all and starve every cleanup that
116
+ # follows. Session close gets longer because it drains several SDK activities.
117
+ _CLEANUP_STEP_TIMEOUT_SECONDS = 8.0
118
+ _SESSION_CLEANUP_TIMEOUT_SECONDS = 15.0
119
+ _BACKGROUND_AUDIO_CLEANUP_TIMEOUT_SECONDS = 5.0
120
+ _NO_CONVERSATION_TIMEOUT_SECONDS = 120.0
121
+ # How long the side that was meant to speak first is given before the simulated person speaks
122
+ # instead. Both sides are voice agents waiting to be addressed, so when the one that placed the call
123
+ # says nothing the call is silence until a deadline discards it, and nothing was learned about
124
+ # either side. Kept well under the timeout above, which is what abandons a call nobody started.
125
+ #
126
+ # Eight seconds: the smallest bound that cannot pre-empt a slow first turn.
127
+ _OPEN_INSTEAD_AFTER_SECONDS = 8.0
128
+ # Frequency and length per kind of mailbox. FULL has no entry: it invites no message.
129
+ _VOICEMAIL_TONE_BY_STYLE: dict[str, tuple[float, float]] = {
130
+ "personal": (1000.0, 0.40),
131
+ "carrier": (1400.0, 0.33),
132
+ "operator": (440.0, 0.52),
133
+ }
134
+ _DEFAULT_VOICEMAIL_STYLE = "personal"
135
+ # Loud enough to be unmistakable against speech, which reaches 15000 to 23000 of 32768.
136
+ _VOICEMAIL_TONE_VOLUME = 0.8
137
+ # A mailbox plays one greeting and then records, so the eight-message conversation floor is
138
+ # unreachable however well the agent behaves, and holding it there errored every voicemail call.
139
+ _VOICEMAIL_MIN_TURN_MESSAGES = 1
140
+ # How long a mailbox records before cutting the line, from the end of the tone or greeting. Without a
141
+ # bound the call runs to the silence watchdog with the agent still talking into a machine.
142
+ _VOICEMAIL_RECORD_SECONDS = 40.0
143
+ # Resamplers into the mixer's rate, one per source rate, kept because ``ratecv`` is stateful.
144
+ _MIXER_RESAMPLERS: dict[tuple[int, int], PCMResampler] = {}
145
+ # How long after the mailbox stops speaking the tone comes. A real system leaves a beat.
146
+ _VOICEMAIL_TONE_GAP_SECONDS = 0.7
147
+ # How long to wait for the mailbox to say anything before giving up on the tone. Bounded so a
148
+ # mailbox that never speaks cannot leave this task pending for the length of the call.
149
+ _VOICEMAIL_TONE_WAIT_SECONDS = 40.0
150
+ # The mixer reinterprets frames at this rate rather than resampling them, so anything published
151
+ # through it must be produced here or it plays at the wrong pitch and length.
152
+ _BACKGROUND_MIXER_RATE = 48000
153
+ # Each web case drives a full voice pipeline (STT/LLM/TTS + LiveKit conns) in one
154
+ # child; too many starve the pod's CPU. This is an OPS CEILING on the
155
+ # config-driven ``max_parallel_cases`` (not a replacement for it) — tune
156
+ # ``ALK_VOICE_MAX_CASE_CONCURRENCY`` to the pod's cores. Caps web cases only.
157
+ _VOICE_MAX_CASE_CONCURRENCY_DEFAULT = 4
158
+
159
+
160
+ def _simulator_participant_identity(persona: Persona, test_case_id: str) -> str:
161
+ """Give repository agents the scenario caller ANI through a standard identity seam.
162
+
163
+ LiveKit token metadata is not exposed consistently across every SDK/agent version. The
164
+ harness therefore uses the identity convention already understood by repository voice
165
+ agents: ``fagi-simulator-phone-<digits>-...``. A persona without a fixture-derived phone
166
+ keeps the legacy anonymous identity.
167
+ """
168
+ definition = persona.persona if isinstance(persona.persona, dict) else {}
169
+ metadata = definition.get("metadata")
170
+ metadata = metadata if isinstance(metadata, dict) else {}
171
+ digits = re.sub(r"\D", "", str(metadata.get("caller_phone") or ""))
172
+ suffix = test_case_id[-12:]
173
+ if 7 <= len(digits) <= 15:
174
+ return f"fagi-simulator-phone-{digits}-{suffix}"
175
+ return f"fagi-simulator-{suffix}"
176
+
177
+
178
+ def _voice_max_case_concurrency() -> int:
179
+ raw = os.environ.get("ALK_VOICE_MAX_CASE_CONCURRENCY", "").strip()
180
+ if not raw:
181
+ return _VOICE_MAX_CASE_CONCURRENCY_DEFAULT
182
+ try:
183
+ value = int(raw)
184
+ except ValueError:
185
+ return _VOICE_MAX_CASE_CONCURRENCY_DEFAULT
186
+ return value if value >= 1 else _VOICE_MAX_CASE_CONCURRENCY_DEFAULT
187
+
188
+
189
+ _silero_vad: Any | None = None
190
+ _silero_vad_guard = threading.Lock()
191
+
192
+
193
+ def _load_silero_vad_sync() -> Any:
194
+ """One shared VAD per process; per-case ``VAD.load()`` ran a synchronous
195
+ model load on the event loop for every concurrent case."""
196
+ global _silero_vad
197
+ with _silero_vad_guard:
198
+ if _silero_vad is None:
199
+ _silero_vad = silero.VAD.load()
200
+ return _silero_vad
201
+
202
+
203
+ @dataclass(frozen=True)
204
+ class _TargetParticipant:
205
+ identity: str
206
+ sid: str
207
+ audio_track_sid: str
208
+ attributes: dict[str, str] = field(default_factory=dict)
209
+
210
+
211
+ @dataclass
212
+ class _CaseOutcome:
213
+ status: TestCaseStatus
214
+ transcript: str = ""
215
+ messages: list[dict[str, str]] = field(default_factory=list)
216
+ failure: SimulationFailure | None = None
217
+ audio_input_path: str | None = None
218
+ audio_output_path: str | None = None
219
+ audio_combined_path: str | None = None
220
+ audio_stereo_path: str | None = None
221
+ metadata: dict[str, object] = field(default_factory=dict)
222
+ evidence: list[EvidenceSourceSummary] = field(default_factory=list)
223
+ provider_artifacts: list[ArtifactManifestEntry] = field(default_factory=list)
224
+
225
+
226
+ def _dispatch_metadata_json(agent_definition) -> str:
227
+ """Metadata for the target agent's LiveKit dispatch.
228
+
229
+ EMPTY by default: a target agent built from a LiveKit template branches on
230
+ ``ctx.job.metadata`` and treats any non-empty payload as an outbound/no-greet
231
+ job, so it never publishes an audio track and readiness times out
232
+ (``agent_unavailable``). Only a target explicitly built to consume dispatch
233
+ metadata sets ``agent_definition.dispatch_metadata``.
234
+ """
235
+ meta = getattr(agent_definition, "dispatch_metadata", None)
236
+ return json.dumps(meta, sort_keys=True) if meta else ""
237
+
238
+
239
+ def _resolve_target_profile(kind: str):
240
+ """Look up the target adapter's profile — the factory that replaced the
241
+ engine's ``transport.kind`` branching. Unknown kinds fail loudly, which is
242
+ what makes it safe to open ``TelephonyTransport.kind`` from a Literal to a
243
+ free string later."""
244
+ from fi.simulate.endpoints.profiles import get_profile
245
+
246
+ profile = get_profile(kind)
247
+ if profile is None:
248
+ raise ValueError(f"unsupported_transport_kind: {kind}")
249
+ return profile
250
+
251
+
252
+ def _simulator_turn_handling(
253
+ *,
254
+ vad: object | None,
255
+ allow_interruptions: bool | None = None,
256
+ min_endpointing_delay: float | None = None,
257
+ max_endpointing_delay: float | None = None,
258
+ ) -> dict[str, object]:
259
+ return {
260
+ "turn_detection": "vad" if vad is not None else "stt",
261
+ # A short delay fires inside a sentence, on a comma or a breath, so the caller treats a pause
262
+ # as the end of the turn, talks over the agent and then repeats itself for want of an answer.
263
+ "endpointing": {
264
+ "mode": "fixed",
265
+ "min_delay": min_endpointing_delay or 0.9,
266
+ "max_delay": max_endpointing_delay or 3.0,
267
+ },
268
+ # A real caller interrupts, but only over something long enough to be worth interrupting.
269
+ "interruption": {
270
+ "enabled": (True if allow_interruptions is None else allow_interruptions),
271
+ "discard_audio_if_uninterruptible": True,
272
+ "min_duration": 0.6,
273
+ },
274
+ "preemptive_generation": {"enabled": True},
275
+ }
276
+
277
+
278
+ class _TestRunnerAgent(Agent):
279
+ def __init__(
280
+ self,
281
+ persona: Persona,
282
+ *,
283
+ min_turn_messages: int = 0,
284
+ **kwargs,
285
+ ):
286
+ turn_handling = kwargs.setdefault(
287
+ "turn_handling",
288
+ _simulator_turn_handling(vad=kwargs.get("vad")),
289
+ )
290
+ super().__init__(**kwargs)
291
+ self._persona = persona
292
+ self._min_turn_messages = min_turn_messages
293
+ self._session_turn_handling = turn_handling
294
+ self._session: AgentSession | None = None
295
+ self._end_requested = asyncio.Event()
296
+ self._end_speech_handle: Any | None = None
297
+ self._usage_collector = metrics.ModelUsageCollector()
298
+
299
+ @function_tool(
300
+ name="endCall",
301
+ # Nothing quotable and nothing English-specific: wording here comes back out as speech.
302
+ description=(
303
+ "Ends the call. Nothing else ends it and no one else ends it for you. "
304
+ "Use it once you have nothing further."
305
+ ),
306
+ )
307
+ async def end_call(self, ctx: RunContext) -> str:
308
+ if self._session is None:
309
+ logger.warning("endCall refused: no session yet")
310
+ return "Continue the conversation before ending the call."
311
+ messages = _session_messages(self._session)
312
+ floor, alternation_required = _turn_requirements(self._min_turn_messages)
313
+ below_floor = len(messages) < floor or (
314
+ alternation_required and not _has_role_alternation(messages)
315
+ )
316
+ if below_floor and _target_has_gone_quiet(messages):
317
+ below_floor = False
318
+ if below_floor:
319
+ # Whether the caller ever reached for this tool, and why it was turned away, is the
320
+ # difference between a simulator that will not hang up and one that was not allowed to.
321
+ logger.warning(
322
+ "endCall refused: %d messages, floor %d, alternating=%s",
323
+ len(messages),
324
+ floor,
325
+ _has_role_alternation(messages),
326
+ )
327
+ # "Not yet" rather than "stop asking", or the caller never retries the tool.
328
+ return (
329
+ f"Not yet: {len(messages)} of {floor} messages so far and both speakers must "
330
+ "have spoken. Keep the conversation going, then call endCall again."
331
+ )
332
+ logger.warning("endCall accepted after %d messages", len(messages))
333
+ # The tool runs inside the same SpeechHandle that carries the model's
334
+ # natural closing sentence. Remember that exact handle before waking
335
+ # the outer runner so it cannot snapshot history in the brief interval
336
+ # before TTS starts and ``session.current_speech`` becomes non-None.
337
+ self._end_speech_handle = ctx.speech_handle
338
+ self._end_requested.set()
339
+ return "Conversation ended."
340
+
341
+ async def wait_for_end_speech(self) -> None:
342
+ if self._end_speech_handle is not None:
343
+ await self._end_speech_handle
344
+
345
+ @property
346
+ def started_session(self) -> AgentSession | None:
347
+ return self._session
348
+
349
+ @property
350
+ def end_requested(self) -> asyncio.Event:
351
+ return self._end_requested
352
+
353
+ @property
354
+ def model_usage(self) -> list[dict[str, object]]:
355
+ return [
356
+ usage.model_dump(mode="json")
357
+ for usage in sorted(
358
+ self._usage_collector.flatten(),
359
+ key=lambda usage: (usage.type, usage.provider, usage.model),
360
+ )
361
+ ]
362
+
363
+ async def start_session(
364
+ self,
365
+ room: rtc.Room,
366
+ *,
367
+ participant_kinds: list | None = None,
368
+ participant_identity: str | None = None,
369
+ ) -> AgentSession:
370
+ session = AgentSession(
371
+ stt=self.stt,
372
+ llm=self.llm,
373
+ tts=self.tts,
374
+ vad=self.vad,
375
+ turn_handling=self._session_turn_handling,
376
+ )
377
+ self._session = session
378
+ session.on(
379
+ "metrics_collected",
380
+ lambda event: self._usage_collector.collect(event.metrics),
381
+ )
382
+ default_kinds = [
383
+ rtc.ParticipantKind.PARTICIPANT_KIND_STANDARD,
384
+ getattr(
385
+ rtc.ParticipantKind,
386
+ "PARTICIPANT_KIND_AGENT",
387
+ rtc.ParticipantKind.PARTICIPANT_KIND_STANDARD,
388
+ ),
389
+ rtc.ParticipantKind.PARTICIPANT_KIND_SIP,
390
+ ]
391
+ room_kwargs: dict = {
392
+ "audio_input": AudioInputOptions(
393
+ pre_connect_audio=False,
394
+ pre_connect_audio_timeout=3.0,
395
+ ),
396
+ # Enabled to build RoomIO's TranscriptSynchronizer, which aligns the
397
+ # spoken transcript to audio playback. On an interruption the
398
+ # recorded turn is then truncated to what was actually said instead
399
+ # of the full LLM text (playback-timing estimate — works with any
400
+ # TTS, unlike use_tts_aligned_transcript which needs word timing our
401
+ # Deepgram/Gemini voices don't emit and would drop the turn). The
402
+ # simulator's transcription is published to the room as a harmless
403
+ # side effect (our target-transcription handler filters by identity).
404
+ "text_output": True,
405
+ "close_on_disconnect": False,
406
+ "delete_room_on_close": False,
407
+ "participant_kinds": participant_kinds or default_kinds,
408
+ }
409
+ if participant_identity:
410
+ room_kwargs["participant_identity"] = participant_identity
411
+ await session.start(
412
+ self,
413
+ room=room,
414
+ room_options=RoomOptions(**room_kwargs),
415
+ )
416
+ await self._maybe_start_background_audio(room, session)
417
+ return session
418
+
419
+ async def _maybe_start_background_audio(
420
+ self, room: "rtc.Room", session: "AgentSession"
421
+ ) -> None:
422
+ """Mix caller-side ambient noise under the simulated caller, if the run asked for it.
423
+
424
+ Off unless HARNESS_BACKGROUND_NOISE names a source: a LiveKit builtin clip name, or an
425
+ http(s) URL to an ambient file. Any failure is swallowed, because a call without ambience is
426
+ preferable to a dropped one.
427
+ """
428
+ source = os.environ.get("HARNESS_BACKGROUND_NOISE", "").strip()
429
+ # A mailbox needs this method for its tone and its recording timer, and FULL has no tone.
430
+ tone_style = _voicemail_tone_style()
431
+ if _answered_by_voicemail() and source:
432
+ # Nothing stands behind a recording, and a room behind one gives the game away.
433
+ logger.info("mailbox answered, so ambience is dropped (noise %r)", source)
434
+ source = ""
435
+ if not source and not tone_style and not _answered_by_voicemail():
436
+ return
437
+
438
+ try:
439
+ # 2.0, not the 0.3 this used to default to. Measured in an isolated two-participant
440
+ # room, the office clip peaks at 119 of 32768 at 0.3, which is below the noise floor of
441
+ # speech near 15000: the ambience played and nobody could hear it. At 2.0 the same clip
442
+ # measures 752 to 789 on real calls, which is audible under a voice without masking it.
443
+ volume = float(os.environ.get("HARNESS_BACKGROUND_NOISE_VOLUME", "2.0"))
444
+ clip_source: Any = None
445
+ if source.startswith(("http://", "https://")):
446
+ clip_source = await asyncio.to_thread(_downloaded_audio, source)
447
+ if not clip_source:
448
+ return
449
+ self._background_noise_file = clip_source
450
+ elif source:
451
+ clip_source = getattr(BuiltinAudioClip, source, None)
452
+ if clip_source is None:
453
+ logger.warning(
454
+ "background audio clip %r is not one LiveKit ships", source
455
+ )
456
+ return
457
+ # A player is created even with no ambience clip, because a mailbox tone needs a
458
+ # published track whether or not this scenario also asked for a room.
459
+ player = (
460
+ BackgroundAudioPlayer(
461
+ ambient_sound=AudioConfig(clip_source, volume=volume)
462
+ )
463
+ if clip_source is not None
464
+ else BackgroundAudioPlayer()
465
+ )
466
+ await player.start(room=room, agent_session=session)
467
+ self._background_player = player
468
+ except Exception:
469
+ logger.warning("background audio not started", exc_info=True)
470
+ return
471
+ # Spoken as its own turn, so the transcript shows what the agent heard.
472
+ recorded = os.environ.get("HARNESS_VOICEMAIL_CLIP", "").strip()
473
+ if recorded.startswith(("http://", "https://")):
474
+ # Catalogue clips are meant to be served from object storage rather than shipped in the
475
+ # image, and a URL cannot be decoded in place.
476
+ recorded = await asyncio.to_thread(_downloaded_audio, recorded) or ""
477
+ said = os.environ.get("HARNESS_VOICEMAIL_CLIP_TRANSCRIPT", "").strip()
478
+ if recorded and said:
479
+ try:
480
+ self._voicemail_greeting = session.say(
481
+ said,
482
+ audio=audio_frames_from_file(recorded),
483
+ allow_interruptions=False,
484
+ )
485
+ except Exception:
486
+ logger.warning("recorded mailbox greeting not played", exc_info=True)
487
+ elif recorded:
488
+ # No words for it, so it cannot be a turn: an invented line would put words in the
489
+ # transcript that the audio never says, and an eval would judge those words.
490
+ logger.warning("mailbox clip has no transcript; playing it without a turn")
491
+ try:
492
+ player.play(AudioConfig(recorded, volume=1.0))
493
+ except Exception:
494
+ logger.warning("recorded mailbox greeting not played", exc_info=True)
495
+ if tone_style:
496
+ self._voicemail_tone_task = asyncio.create_task(
497
+ self._play_voicemail_tone(tone_style, session)
498
+ )
499
+ if _answered_by_voicemail():
500
+ self._mailbox_close_task = asyncio.create_task(
501
+ self._close_mailbox_after_recording()
502
+ )
503
+
504
+ _voicemail_greeting: Any = None
505
+ _voicemail_tone_task: Any = None
506
+ _mailbox_close_task: Any = None
507
+
508
+ async def _close_mailbox_after_recording(self) -> None:
509
+ """Stop recording and cut the line, the way a mailbox does.
510
+
511
+ A mailbox is not a party to the call. It never says goodbye, it never asks whether anybody is
512
+ there, and it does not wait: it records for as long as it records and then hangs up. Nothing
513
+ here was ending these calls, so they ran to the silence watchdog with the agent talking into
514
+ a machine long after it had left its message.
515
+
516
+ The clock starts once the greeting and the tone are done where there are either, and at call
517
+ start otherwise, which is the FULL mailbox: it invites no message, so the window it gets is
518
+ generous rather than precise.
519
+ """
520
+ try:
521
+ if self._voicemail_greeting is not None:
522
+ await self._voicemail_greeting
523
+ tone = self._voicemail_tone_task
524
+ if tone is not None:
525
+ try:
526
+ await tone
527
+ except Exception:
528
+ # The tone failing is not a reason to record for ever.
529
+ logger.warning(
530
+ "mailbox tone failed before the recording timer", exc_info=True
531
+ )
532
+ await asyncio.sleep(_VOICEMAIL_RECORD_SECONDS)
533
+ logger.info(
534
+ "mailbox stopped recording after %ss and cut the line",
535
+ _VOICEMAIL_RECORD_SECONDS,
536
+ )
537
+ self._end_requested.set()
538
+ except asyncio.CancelledError:
539
+ raise
540
+ except Exception:
541
+ # A mailbox that fails to hang up leaves the watchdog to end the call, which is the
542
+ # behaviour this replaces rather than a new failure.
543
+ logger.warning("mailbox recording timer failed", exc_info=True)
544
+
545
+ async def _play_voicemail_tone(self, style: str, session: "AgentSession") -> None:
546
+ """Play the tone a mailbox plays once its greeting has finished.
547
+
548
+ The greeting comes either from a catalogue recording or from this session speaking the
549
+ persona's opening line. The tone is the part neither can carry, and without it a greeting that
550
+ says "leave a message after the tone" asks the agent to wait for something that never comes.
551
+
552
+ Generated rather than fetched: a sine burst is a sine burst, and an asset would be a
553
+ download, a licence and a catalogue for something twelve lines of arithmetic produce. Handed
554
+ over eagerly rather than lazily, since the player consumes the iterator inside its mixer
555
+ task and a failure there can take the ambience down with it.
556
+ """
557
+ shape = _VOICEMAIL_TONE_BY_STYLE.get(style)
558
+ if shape is None:
559
+ return
560
+ hz, seconds = shape
561
+ try:
562
+ # After the greeting, not at a guessed offset: wait for the mailbox's own first
563
+ # committed turn, which is this session's assistant role, then leave a beat.
564
+ recorded = os.environ.get("HARNESS_VOICEMAIL_CLIP", "").strip()
565
+ if recorded:
566
+ # Awaiting the greeting's own handle is exact where a duration would be a guess.
567
+ if self._voicemail_greeting is not None:
568
+ await self._voicemail_greeting
569
+ else:
570
+ loop = asyncio.get_running_loop()
571
+ deadline = loop.time() + _VOICEMAIL_TONE_WAIT_SECONDS
572
+ while loop.time() < deadline:
573
+ if any(
574
+ message["content"]
575
+ for message in _session_messages(session)
576
+ if message["role"] == "assistant"
577
+ ):
578
+ break
579
+ await asyncio.sleep(0.2)
580
+ else:
581
+ logger.warning(
582
+ "mailbox said nothing in %ss; no tone",
583
+ _VOICEMAIL_TONE_WAIT_SECONDS,
584
+ )
585
+ return
586
+ await asyncio.sleep(_VOICEMAIL_TONE_GAP_SECONDS)
587
+ player = getattr(self, "_background_player", None)
588
+ if player is None:
589
+ return
590
+ # A recording that ends with its own tone replaces this one rather than preceding it.
591
+ # Two beeps is worse than one, and the catalogue records which clips carry theirs.
592
+ if os.environ.get("HARNESS_VOICEMAIL_CLIP_HAS_TONE", "").strip() == "1":
593
+ return
594
+ spoken = [_tone_frame(hz, seconds)]
595
+
596
+ async def frames() -> Any:
597
+ for frame in spoken:
598
+ yield frame
599
+
600
+ player.play(AudioConfig(frames(), volume=_VOICEMAIL_TONE_VOLUME))
601
+ except asyncio.CancelledError:
602
+ raise
603
+ except Exception:
604
+ # A mailbox without its tone is a weaker test, never a failed call.
605
+ logger.warning("voicemail tone not played", exc_info=True)
606
+
607
+ async def _stop_background_audio(self) -> None:
608
+ """Close the ambience player and remove any clip downloaded for it.
609
+
610
+ Without this the mixer task, its audio source and the published track outlive the call,
611
+ and a suite leaks one of each (plus a temp file) per scenario.
612
+ """
613
+ for name in ("_voicemail_tone_task", "_mailbox_close_task"):
614
+ pending = getattr(self, name, None)
615
+ if pending is not None:
616
+ setattr(self, name, None)
617
+ if not pending.done():
618
+ pending.cancel()
619
+ player = getattr(self, "_background_player", None)
620
+ if player is not None:
621
+ self._background_player = None
622
+ try:
623
+ await player.aclose()
624
+ except Exception:
625
+ logger.warning("background audio not closed cleanly", exc_info=True)
626
+ downloaded = getattr(self, "_background_noise_file", None)
627
+ if downloaded:
628
+ self._background_noise_file = None
629
+ try:
630
+ Path(downloaded).unlink(missing_ok=True)
631
+ except OSError:
632
+ logger.warning("background audio clip not removed: %s", downloaded)
633
+
634
+ def open_conversation(self) -> None:
635
+ if self._session is None:
636
+ raise RuntimeError("simulator_session_not_started")
637
+ if self._voicemail_greeting is not None:
638
+ # A recording has already greeted, and a mailbox does not greet twice: a spoken line on
639
+ # top of the clip is one mailbox answering in two voices.
640
+ return
641
+ initial_message = self._persona.persona.get("initial_message")
642
+ if isinstance(initial_message, str) and initial_message.strip():
643
+ self._session.say(initial_message.strip())
644
+ return
645
+ self._session.generate_reply()
646
+
647
+ _mailbox_greeted: bool = False
648
+
649
+ async def llm_node(self, chat_ctx, tools, model_settings):
650
+ """A mailbox speaks once and then never again, counted here rather than asked of the model.
651
+
652
+ One turn is allowed, not none, because a mailbox without a recording greets through this path.
653
+ Where a recording has already greeted, no turn is allowed at all.
654
+ """
655
+ if _answered_by_voicemail():
656
+ if self._mailbox_greeted or self._voicemail_greeting is not None:
657
+ return
658
+ self._mailbox_greeted = True
659
+ async for chunk in super().llm_node(chat_ctx, tools, model_settings):
660
+ yield chunk
661
+
662
+ async def transcription_node(
663
+ self,
664
+ text: AsyncIterable[str | TimedString],
665
+ model_settings: ModelSettings,
666
+ ):
667
+ async for chunk in text:
668
+ logger.debug(
669
+ "Simulator transcription chunk",
670
+ extra={"timed": isinstance(chunk, TimedString)},
671
+ )
672
+ yield chunk
673
+
674
+
675
+ class LiveKitEngine(BaseEngine):
676
+ async def run(
677
+ self,
678
+ agent_definition: AgentDefinition | None = None,
679
+ livekit_runtime: LiveKitSimulatorRuntime | None = None,
680
+ scenario: Scenario | None = None,
681
+ simulator: SimulatorAgentDefinition | None = None,
682
+ num_scenarios: int = 1,
683
+ topic: str | None = None,
684
+ record_audio: bool = False,
685
+ recorder_sample_rate: int = 8000,
686
+ recorder_join_delay: float = 0.2,
687
+ min_turn_messages: int = 8,
688
+ max_seconds: float = 45.0,
689
+ connect_timeout: float = 15.0,
690
+ readiness_timeout: float = 30.0,
691
+ cleanup_timeout: float = 30.0,
692
+ conversation_direction: str = "simulator_first",
693
+ agent_first_silence_timeout_seconds: float = 120.0,
694
+ recording_root: str | Path = "recordings",
695
+ recording_case_directory: str | Path | None = None,
696
+ run_id: str | None = None,
697
+ max_concurrency: int = 1,
698
+ on_case_complete: Callable[[int, TestCaseResult], Awaitable[None]]
699
+ | None = None,
700
+ on_case_start: Callable[[int], Awaitable[None]] | None = None,
701
+ **kwargs,
702
+ ) -> TestReport:
703
+ if agent_definition is None:
704
+ raise ValueError("LiveKitEngine requires 'agent_definition'.")
705
+ runtime = _resolve_livekit_runtime(agent_definition, livekit_runtime)
706
+ if conversation_direction not in {"simulator_first", "agent_first"}:
707
+ raise ValueError(
708
+ "conversation_direction must be simulator_first or agent_first"
709
+ )
710
+ if agent_first_silence_timeout_seconds <= 0:
711
+ raise ValueError("agent_first_silence_timeout_seconds must be positive")
712
+ if scenario is None:
713
+ generator = ScenarioGenerator(
714
+ agent_definition,
715
+ llm_config=(
716
+ simulator.llm
717
+ if simulator is not None
718
+ else _default_simulator_llm_config()
719
+ ),
720
+ )
721
+ if topic is None:
722
+ simulator_context = (
723
+ simulator.instructions
724
+ if simulator and simulator.instructions
725
+ else ""
726
+ )
727
+ topic = (
728
+ simulator_context
729
+ or agent_definition.system_prompt
730
+ or "customer support scenarios"
731
+ ).strip()
732
+ personas = await generator.generate(
733
+ topic=topic,
734
+ num_personas=num_scenarios,
735
+ )
736
+ scenario = Scenario(name="Generated Scenario", dataset=personas)
737
+ transport = agent_definition.transport or TelephonyTransport()
738
+ profile = _resolve_target_profile(transport.kind)
739
+ # Computed before the room-name checks below: a serial (concurrency-1)
740
+ # run is what makes reusing one fixed room across cases safe, so the
741
+ # verbatim check needs this value, not just the dataset size.
742
+ # Cases run concurrently up to ``max_concurrency`` (bounded by the
743
+ # customer agent's own session capacity). SIP legs stay serial: a run
744
+ # leases a single DID, so overlapping calls would collide. Ask the
745
+ # resolved profile rather than re-branching on ``transport.kind``.
746
+ case_concurrency = (
747
+ 1
748
+ if profile.is_sip
749
+ else max(
750
+ 1,
751
+ min(
752
+ int(max_concurrency or 1),
753
+ _voice_max_case_concurrency(),
754
+ len(scenario.dataset),
755
+ ),
756
+ )
757
+ )
758
+ if (
759
+ runtime.room_name_verbatim
760
+ and len(scenario.dataset) != 1
761
+ and case_concurrency != 1
762
+ ):
763
+ raise ValueError(
764
+ "room_name_verbatim_requires_serial_cases: a fixed room hosts "
765
+ "one case at a time; run with max_concurrency=1"
766
+ )
767
+ if (
768
+ runtime.room_mode == "external"
769
+ and len(scenario.dataset) > 1
770
+ and not _has_room_template(runtime.room_name)
771
+ ):
772
+ raise ValueError(
773
+ "external_room_template_required: concurrent-safe multi-case runs "
774
+ "need {run_id}, {test_case_id}, or {index} in room_name"
775
+ )
776
+ if not profile.uses_external_room and runtime.room_mode != "managed":
777
+ raise ValueError("managed_transport_requires_managed_room")
778
+ if (
779
+ profile.receives_inbound_call
780
+ and len(scenario.dataset) > 1
781
+ and not _has_room_template(runtime.room_name)
782
+ and not runtime.room_name_verbatim
783
+ ):
784
+ raise ValueError(
785
+ "sip_inbound_room_template_required: multi-case inbound runs "
786
+ "need {run_id} or {test_case_id} in room_name"
787
+ )
788
+ cleanup_timeout = min(cleanup_timeout, _MAX_CLEANUP_TIMEOUT_SECONDS)
789
+ current_run_id = run_id or new_run_id()
790
+ if recording_case_directory is not None and len(scenario.dataset) != 1:
791
+ raise ValueError(
792
+ "recording_case_directory requires a single-persona scenario"
793
+ )
794
+ invocation_id = uuid4().hex[:12]
795
+ report = TestReport()
796
+
797
+ case_semaphore = asyncio.Semaphore(case_concurrency)
798
+
799
+ async def _run_case(index: int, persona: Persona) -> TestCaseResult:
800
+ async with case_semaphore:
801
+ # Mark this case's row ONGOING the moment it claims a concurrency
802
+ # slot — cases still queued behind the semaphore stay PENDING.
803
+ # Best-effort and engine-agnostic; a failed ping never fails the
804
+ # case (the backend gates the update on PENDING).
805
+ if on_case_start is not None:
806
+ try:
807
+ await on_case_start(index)
808
+ except Exception as exc: # noqa: BLE001
809
+ logger.error(
810
+ "voice case start callback failed",
811
+ exc_info=redacted_exc_info(exc),
812
+ extra={
813
+ "run_id": current_run_id,
814
+ "case_index": index,
815
+ },
816
+ )
817
+ persona_ref = persona.version or persona.content_hash()
818
+ test_case_id = derive_test_case_id(
819
+ current_run_id,
820
+ persona_ref,
821
+ index,
822
+ )
823
+ room_name = _resolve_room_name(
824
+ runtime,
825
+ run_id=current_run_id,
826
+ test_case_id=test_case_id,
827
+ index=index,
828
+ invocation_id=invocation_id,
829
+ )
830
+ case_directory = (
831
+ Path(recording_case_directory)
832
+ if recording_case_directory is not None
833
+ else Path(recording_root) / current_run_id / test_case_id
834
+ )
835
+ try:
836
+ outcome = await self._run_single_test_case(
837
+ agent_definition,
838
+ runtime,
839
+ persona,
840
+ simulator,
841
+ run_id=current_run_id,
842
+ test_case_id=test_case_id,
843
+ invocation_id=invocation_id,
844
+ room_name=room_name,
845
+ case_directory=case_directory,
846
+ record_audio=record_audio,
847
+ recorder_sample_rate=recorder_sample_rate,
848
+ recorder_join_delay=recorder_join_delay,
849
+ min_turn_messages=min_turn_messages,
850
+ max_seconds=max_seconds,
851
+ connect_timeout=connect_timeout,
852
+ readiness_timeout=readiness_timeout,
853
+ cleanup_timeout=cleanup_timeout,
854
+ conversation_direction=conversation_direction,
855
+ agent_first_silence_timeout_seconds=agent_first_silence_timeout_seconds,
856
+ )
857
+ except asyncio.CancelledError:
858
+ raise
859
+ except Exception as exc: # noqa: BLE001
860
+ # One case crashing must neither sink the batch nor leave a
861
+ # hole that shifts the positional result-to-CallExecution
862
+ # mapping — emit a dense failed case in its slot.
863
+ logger.error(
864
+ "voice case crashed",
865
+ exc_info=redacted_exc_info(exc),
866
+ extra={
867
+ "run_id": current_run_id,
868
+ "test_case_id": test_case_id,
869
+ },
870
+ )
871
+ failure = SimulationFailure(
872
+ stage=FailureStage.RUNNING,
873
+ code="case_execution_error",
874
+ message=f"{type(exc).__name__}: {exc}",
875
+ retryable=False,
876
+ )
877
+ result = TestCaseResult(
878
+ persona=persona,
879
+ transcript="",
880
+ messages=[],
881
+ metadata={
882
+ "engine": "livekit",
883
+ "run_id": current_run_id,
884
+ "test_case_id": test_case_id,
885
+ "invocation_id": invocation_id,
886
+ "status": TestCaseStatus.FAILED.value,
887
+ "room_name": room_name,
888
+ "room_mode": runtime.room_mode,
889
+ "failure": failure.model_dump(
890
+ mode="json", exclude_none=True
891
+ ),
892
+ },
893
+ )
894
+ else:
895
+ metadata = {
896
+ "engine": "livekit",
897
+ "run_id": current_run_id,
898
+ "test_case_id": test_case_id,
899
+ "invocation_id": invocation_id,
900
+ "status": outcome.status.value,
901
+ "room_name": room_name,
902
+ "room_mode": runtime.room_mode,
903
+ **outcome.metadata,
904
+ }
905
+ if outcome.failure is not None:
906
+ metadata["failure"] = outcome.failure.model_dump(
907
+ mode="json",
908
+ exclude_none=True,
909
+ )
910
+ if outcome.evidence:
911
+ metadata["evidence"] = [
912
+ item.model_dump(mode="json", exclude_none=True)
913
+ for item in outcome.evidence
914
+ ]
915
+ if outcome.provider_artifacts:
916
+ metadata["provider_artifacts"] = [
917
+ entry.model_dump(mode="json", exclude_none=True)
918
+ for entry in outcome.provider_artifacts
919
+ ]
920
+ result = TestCaseResult(
921
+ persona=persona,
922
+ transcript=outcome.transcript,
923
+ messages=outcome.messages,
924
+ metadata=metadata,
925
+ audio_input_path=outcome.audio_input_path,
926
+ audio_output_path=outcome.audio_output_path,
927
+ audio_combined_path=outcome.audio_combined_path,
928
+ audio_stereo_path=outcome.audio_stereo_path,
929
+ )
930
+
931
+ # Stream the finished case AFTER releasing the semaphore — a slow
932
+ # result PATCH (recording upload) must not hold a concurrency slot.
933
+ # A streaming error never fails the case; finalize reconciles it.
934
+ if on_case_complete is not None:
935
+ try:
936
+ await on_case_complete(index, result)
937
+ except Exception as exc: # noqa: BLE001
938
+ logger.error(
939
+ "voice case stream callback failed",
940
+ exc_info=redacted_exc_info(exc),
941
+ extra={
942
+ "run_id": current_run_id,
943
+ "test_case_id": test_case_id,
944
+ },
945
+ )
946
+ return result
947
+
948
+ # ``gather`` preserves argument order regardless of completion order, so
949
+ # ``report.results`` stays in dataset order — the positional contract the
950
+ # FutureAGI sink relies on to map results to pre-allocated CallExecutions.
951
+ results = await asyncio.gather(
952
+ *(
953
+ _run_case(index, persona)
954
+ for index, persona in enumerate(scenario.dataset)
955
+ )
956
+ )
957
+ report.results.extend(results)
958
+ return report
959
+
960
+ async def _run_single_test_case(
961
+ self,
962
+ agent_definition: AgentDefinition,
963
+ runtime: LiveKitSimulatorRuntime,
964
+ persona: Persona,
965
+ simulator: SimulatorAgentDefinition | None,
966
+ *,
967
+ run_id: str,
968
+ test_case_id: str,
969
+ invocation_id: str,
970
+ room_name: str,
971
+ case_directory: Path,
972
+ record_audio: bool,
973
+ recorder_sample_rate: int,
974
+ recorder_join_delay: float,
975
+ min_turn_messages: int,
976
+ max_seconds: float,
977
+ connect_timeout: float,
978
+ readiness_timeout: float,
979
+ cleanup_timeout: float,
980
+ conversation_direction: str,
981
+ agent_first_silence_timeout_seconds: float,
982
+ ) -> _CaseOutcome:
983
+ # Teardown is a run of independent steps that each used to take the full
984
+ # ``cleanup_timeout``. Ten of them at up to sixty seconds is six hundred seconds of
985
+ # cleanup against a run budget of five hundred and seventy, so one slow teardown spent
986
+ # the whole budget and the case was discarded as a timeout with its conversation already
987
+ # finished. Share one deadline across the run instead, started at the first cleanup that
988
+ # actually waits, so every path gets the same bound however it got there.
989
+ _cleanup_started: list[float] = []
990
+
991
+ def _cleanup_budget(
992
+ cap: float = _CLEANUP_STEP_TIMEOUT_SECONDS,
993
+ ) -> float:
994
+ if not _cleanup_started:
995
+ _cleanup_started.append(time.monotonic())
996
+ spent = time.monotonic() - _cleanup_started[0]
997
+ remaining = max(0.1, cleanup_timeout - spent)
998
+ return min(cap, remaining)
999
+
1000
+ api_key = os.environ.get(runtime.api_key_env)
1001
+ api_secret = os.environ.get(runtime.api_secret_env)
1002
+ if not api_key or not api_secret:
1003
+ return _failure_outcome(
1004
+ TestCaseStatus.FAILED,
1005
+ FailureStage.PREPARING,
1006
+ "livekit_credentials_missing",
1007
+ f"{runtime.api_key_env} and {runtime.api_secret_env} are required",
1008
+ )
1009
+ simulator_identity = _simulator_participant_identity(persona, test_case_id)
1010
+ recorder_identity = f"fagi-recorder-{test_case_id[-12:]}"
1011
+ room = rtc.Room()
1012
+ models: LiveKitModels | None = None
1013
+ recorder: RoomRecorder | None = None
1014
+ customer_agent: _TestRunnerAgent | None = None
1015
+ session: AgentSession | None = None
1016
+ api_client: api.LiveKitAPI | None = None
1017
+ target: _TargetParticipant | None = None
1018
+ target_transcription_mode = False
1019
+ target_transcription_handler_registered = False
1020
+ target_transcription_tasks: set[asyncio.Task[None]] = set()
1021
+ # Set the moment the conversation ends. A target transcription stream
1022
+ # still in flight at that point is the target's final utterance; it must
1023
+ # be recorded into the transcript, but WITHOUT triggering another
1024
+ # simulator reply (the call is over).
1025
+ conversation_ended = asyncio.Event()
1026
+ # Every target utterance, captured straight off its transcription stream
1027
+ # independently of the simulator session. Once the session starts
1028
+ # draining it rejects new input ("speech scheduling is paused"), so a
1029
+ # target closing delivered after the simulator is done never reaches the
1030
+ # chat context. These are merged into the report so the trailing target
1031
+ # turn is never lost.
1032
+ captured_target_turns: list[dict[str, Any]] = []
1033
+ # agent_first (target greets first): the target can publish its greeting
1034
+ # transcription before the main handler is registered post-readiness, and
1035
+ # the LiveKit client DROPS a text-stream header that arrives with no
1036
+ # handler. So for managed external-room agent_first we register an early
1037
+ # buffer handler right after connect, defer the target dispatch until the
1038
+ # buffer is live, and drain the buffered streams through the (unchanged,
1039
+ # unconditional) main handler once the target is selected.
1040
+ pending_target_transcriptions: list[tuple["rtc.TextStreamReader", str]] = []
1041
+ target_dispatch_deferred = False
1042
+ _MAX_BUFFERED_TARGET_STREAMS = 16
1043
+ managed_room_owned = runtime.room_mode == "managed"
1044
+ room_connected = False
1045
+ cleanup_errors: list[str] = []
1046
+ outcome: _CaseOutcome | None = None
1047
+ sip_dispatch_rule_id: str | None = None
1048
+ sip_dispatch_rule_created = False
1049
+ call_originator: CallOriginator | None = None
1050
+ provider_call_id: str | None = None
1051
+ provider_termination_source: str | None = None
1052
+ reconciled_call_ids: list[str] = []
1053
+ caller_verification: str | None = None
1054
+ audio_bridge: LiveKitAudioBridge | None = None
1055
+ bridge_task: asyncio.Task[None] | None = None
1056
+ case_started_at = datetime.now(timezone.utc)
1057
+ transport = agent_definition.transport or TelephonyTransport()
1058
+ profile = _resolve_target_profile(transport.kind)
1059
+ provider_target = agent_definition.target
1060
+ effective_target_identity = agent_definition.target_participant_identity
1061
+ effective_readiness_timeout = (
1062
+ transport.readiness_timeout_seconds
1063
+ if profile.receives_inbound_call
1064
+ and transport.readiness_timeout_seconds is not None
1065
+ else readiness_timeout
1066
+ )
1067
+ sip_answer_timeout = transport.answer_timeout_seconds or max(
1068
+ connect_timeout, 60.0
1069
+ )
1070
+
1071
+ def buffer_target_transcription(
1072
+ reader: "rtc.TextStreamReader",
1073
+ participant_identity: str,
1074
+ ) -> None:
1075
+ # Early (pre-readiness) handler for managed external-room agent_first.
1076
+ # ``target.audio_track_sid`` isn't known yet, so attribute by the same
1077
+ # exclusion ``_wait_for_target_audio`` uses — anything that is not the
1078
+ # simulator or recorder is target-worthy. The main handler re-applies
1079
+ # the strict track/identity filter when draining, so over-buffering is
1080
+ # safe.
1081
+ nonlocal target_transcription_mode
1082
+ pid = str(participant_identity)
1083
+ if pid in (simulator_identity, recorder_identity):
1084
+ return
1085
+ if len(pending_target_transcriptions) >= _MAX_BUFFERED_TARGET_STREAMS:
1086
+ logger.warning(
1087
+ "target_transcription_buffer_full: dropping stream (buffered=%d)",
1088
+ len(pending_target_transcriptions),
1089
+ )
1090
+ return
1091
+ pending_target_transcriptions.append((reader, pid))
1092
+ # Kill the duplicate-response race at the source: the target is
1093
+ # speaking, so disable the simulator's STT now — otherwise STT would
1094
+ # also transcribe the greeting and emit a second, duplicate reply.
1095
+ # Dispatch is deferred until after ``session.start()``, so a buffered
1096
+ # stream implies a live session.
1097
+ if not target_transcription_mode and session is not None:
1098
+ session.input.set_audio_enabled(False)
1099
+ session.clear_user_turn()
1100
+ target_transcription_mode = True
1101
+
1102
+ try:
1103
+ if managed_room_owned:
1104
+ api_client = api.LiveKitAPI(
1105
+ _api_url(str(runtime.url)),
1106
+ api_key,
1107
+ api_secret,
1108
+ )
1109
+ if not profile.places_outbound_call:
1110
+ # The pool's dispatch rule stays live for the whole run, so a
1111
+ # stray inbound call can re-create this room at any moment
1112
+ # between cases — drain it before trusting it as ours.
1113
+ if runtime.room_name_verbatim:
1114
+ try:
1115
+ polls = await asyncio.wait_for(
1116
+ _ensure_room_absent(api_client, room_name),
1117
+ timeout=connect_timeout,
1118
+ )
1119
+ logger.info(
1120
+ "leased room drained",
1121
+ extra={
1122
+ "run_id": run_id,
1123
+ "test_case_id": test_case_id,
1124
+ "room_name": room_name,
1125
+ "polls": polls,
1126
+ },
1127
+ )
1128
+ except asyncio.TimeoutError:
1129
+ outcome = _failure_outcome(
1130
+ TestCaseStatus.TIMED_OUT,
1131
+ FailureStage.PREPARING,
1132
+ "livekit_room_drain_timeout",
1133
+ "The leased simulator room was still occupied when its drain deadline passed",
1134
+ retryable=True,
1135
+ )
1136
+ except Exception as exc: # noqa: BLE001
1137
+ logger.warning(
1138
+ "leased room drain failed",
1139
+ exc_info=redacted_exc_info(exc),
1140
+ extra={
1141
+ "run_id": run_id,
1142
+ "test_case_id": test_case_id,
1143
+ "room_name": room_name,
1144
+ },
1145
+ )
1146
+ outcome = _failure_outcome(
1147
+ TestCaseStatus.FAILED,
1148
+ FailureStage.PREPARING,
1149
+ "livekit_room_drain_failed",
1150
+ "Could not clear the leased simulator room before the call",
1151
+ details=_safe_provider_error_details(
1152
+ exc, operation="room_drain"
1153
+ ),
1154
+ )
1155
+ if outcome is None:
1156
+ try:
1157
+ await asyncio.wait_for(
1158
+ api_client.room.create_room(
1159
+ api.CreateRoomRequest(name=room_name)
1160
+ ),
1161
+ timeout=connect_timeout,
1162
+ )
1163
+ except asyncio.TimeoutError:
1164
+ outcome = _failure_outcome(
1165
+ TestCaseStatus.TIMED_OUT,
1166
+ FailureStage.PREPARING,
1167
+ "livekit_room_create_timeout",
1168
+ "LiveKit room creation exceeded its deadline",
1169
+ retryable=True,
1170
+ )
1171
+ except Exception as exc:
1172
+ logger.warning(
1173
+ "LiveKit room creation failed",
1174
+ exc_info=redacted_exc_info(exc),
1175
+ extra={
1176
+ "run_id": run_id,
1177
+ "test_case_id": test_case_id,
1178
+ "room_name": room_name,
1179
+ },
1180
+ )
1181
+ outcome = _failure_outcome(
1182
+ TestCaseStatus.FAILED,
1183
+ FailureStage.PREPARING,
1184
+ "livekit_room_create_failed",
1185
+ "Failed to create the LiveKit room",
1186
+ details=_safe_provider_error_details(
1187
+ exc, operation="room_create"
1188
+ ),
1189
+ )
1190
+ if outcome is None and profile.uses_external_room:
1191
+ # Defer the target dispatch until AFTER the early buffer
1192
+ # handler is registered and the session is live (both
1193
+ # directions): a native target may greet the moment it
1194
+ # joins even when the simulator is meant to open, and the
1195
+ # LiveKit client drops a text-stream header that arrives
1196
+ # with no handler — the greeting would be lost.
1197
+ target_dispatch_deferred = True
1198
+ elif outcome is None and profile.receives_inbound_call:
1199
+ try:
1200
+ (
1201
+ sip_dispatch_rule_id,
1202
+ sip_dispatch_rule_created,
1203
+ ) = await asyncio.wait_for(
1204
+ _ensure_sip_inbound_dispatch(
1205
+ api_client,
1206
+ transport=transport,
1207
+ room_name=room_name,
1208
+ ),
1209
+ timeout=connect_timeout,
1210
+ )
1211
+ except asyncio.TimeoutError:
1212
+ outcome = _failure_outcome(
1213
+ TestCaseStatus.TIMED_OUT,
1214
+ FailureStage.PREPARING,
1215
+ "sip_inbound_dispatch_timeout",
1216
+ "SIP inbound dispatch provisioning exceeded its deadline",
1217
+ retryable=True,
1218
+ )
1219
+ except Exception as exc:
1220
+ logger.warning(
1221
+ "SIP inbound dispatch provisioning failed",
1222
+ exc_info=redacted_exc_info(exc),
1223
+ extra={
1224
+ "run_id": run_id,
1225
+ "test_case_id": test_case_id,
1226
+ "room_name": room_name,
1227
+ },
1228
+ )
1229
+ outcome = _failure_outcome(
1230
+ TestCaseStatus.FAILED,
1231
+ FailureStage.PREPARING,
1232
+ "sip_inbound_dispatch_failed",
1233
+ "Failed to provision SIP inbound dispatch",
1234
+ details=_safe_provider_error_details(
1235
+ exc, operation="sip_dispatch"
1236
+ ),
1237
+ )
1238
+ if outcome is not None:
1239
+ return outcome
1240
+ # The target resolves who is calling from participant attributes or metadata.
1241
+ # Without the persona's number every scenario looks like the same demo rider and
1242
+ # the agent looks up the wrong account, which reads as an agent bug.
1243
+ caller_phone = str(
1244
+ (persona.persona.get("metadata") or {}).get("caller_phone") or ""
1245
+ ).strip()
1246
+ builder = (
1247
+ AccessToken(api_key, api_secret)
1248
+ .with_identity(simulator_identity)
1249
+ .with_grants(VideoGrants(room_join=True, room=room_name))
1250
+ )
1251
+ if caller_phone:
1252
+ builder = builder.with_attributes(
1253
+ {"harness.callerPhone": caller_phone}
1254
+ ).with_metadata(json.dumps({"caller_phone": caller_phone}))
1255
+ token = builder.to_jwt()
1256
+ await asyncio.wait_for(
1257
+ room.connect(str(runtime.url), token),
1258
+ timeout=connect_timeout,
1259
+ )
1260
+ room_connected = True
1261
+ if profile.uses_external_room:
1262
+ # Buffer any target greeting that arrives before readiness; the
1263
+ # LiveKit client drops a text-stream header with no handler.
1264
+ room.register_text_stream_handler(
1265
+ TOPIC_TRANSCRIPTION, buffer_target_transcription
1266
+ )
1267
+ target_transcription_handler_registered = True
1268
+ if record_audio:
1269
+ recorder = RoomRecorder(
1270
+ url=str(runtime.url),
1271
+ api_key=api_key,
1272
+ api_secret=api_secret,
1273
+ room_name=room_name,
1274
+ identity=recorder_identity,
1275
+ sample_rate=recorder_sample_rate,
1276
+ output_dir=case_directory / "audio",
1277
+ join_delay_s=recorder_join_delay,
1278
+ )
1279
+ await asyncio.wait_for(
1280
+ recorder.start(),
1281
+ timeout=connect_timeout,
1282
+ )
1283
+ customer_agent, models = await self._create_customer_agent(
1284
+ persona,
1285
+ simulator,
1286
+ # The AGENT's direction; it picks which half of the role block the caller gets.
1287
+ call_type=(
1288
+ "outbound"
1289
+ if os.environ.get("HARNESS_CALL_DIRECTION", "").strip().lower()
1290
+ == "outbound"
1291
+ else "inbound"
1292
+ ),
1293
+ # `name` is an identity for dispatch, not a label for the caller to hear.
1294
+ agent_name=agent_definition.description,
1295
+ min_turn_messages=min_turn_messages,
1296
+ )
1297
+ setup = getattr(self, "_last_simulator_setup", {}) or {}
1298
+ _record_simulator_setup(
1299
+ case_directory,
1300
+ persona=persona,
1301
+ instructions=setup.get("instructions", ""),
1302
+ llm_config=setup.get("llm_config"),
1303
+ stt_config=setup.get("stt_config"),
1304
+ tts_config=setup.get("tts_config"),
1305
+ extra={
1306
+ "room_name": room_name,
1307
+ "agent_name": agent_definition.name,
1308
+ "test_case_id": test_case_id,
1309
+ "run_id": run_id,
1310
+ "conversation_direction": conversation_direction,
1311
+ "allow_interruptions": setup.get("allow_interruptions"),
1312
+ "min_endpointing_delay": setup.get("min_endpointing_delay"),
1313
+ "max_endpointing_delay": setup.get("max_endpointing_delay"),
1314
+ "use_tts_aligned_transcript": setup.get(
1315
+ "use_tts_aligned_transcript"
1316
+ ),
1317
+ },
1318
+ )
1319
+ sip_participant_identity: str | None = None
1320
+ bridge_identity: str | None = None
1321
+ if profile.places_outbound_call:
1322
+ identity_template = (
1323
+ transport.participant_identity
1324
+ or "sip-caller-{invocation_id}-{test_case_id}"
1325
+ )
1326
+ sip_participant_identity = identity_template.format(
1327
+ test_case_id=test_case_id,
1328
+ run_id=run_id,
1329
+ invocation_id=invocation_id,
1330
+ )
1331
+ if effective_target_identity is None:
1332
+ effective_target_identity = sip_participant_identity
1333
+ elif profile.uses_web_audio_bridge:
1334
+ bridge_identity = (
1335
+ f"fagi-{profile.bridge_provider}-bridge-{test_case_id[-12:]}"
1336
+ )
1337
+ effective_target_identity = bridge_identity
1338
+ session_participant_kinds = None
1339
+ session_participant_identity: str | None = None
1340
+ if profile.joins_as_sip_participant:
1341
+ session_participant_kinds = [rtc.ParticipantKind.PARTICIPANT_KIND_SIP]
1342
+ session_participant_identity = (
1343
+ effective_target_identity or sip_participant_identity
1344
+ )
1345
+ session = await asyncio.wait_for(
1346
+ customer_agent.start_session(
1347
+ room,
1348
+ participant_kinds=session_participant_kinds,
1349
+ participant_identity=session_participant_identity,
1350
+ ),
1351
+ timeout=connect_timeout,
1352
+ )
1353
+ if target_dispatch_deferred:
1354
+ # Session + early buffer handler are live; now dispatch the target
1355
+ # so its greeting stream is captured, not dropped.
1356
+ assert api_client is not None
1357
+ dispatch_agent_name = (
1358
+ agent_definition.agent_name or agent_definition.name
1359
+ )
1360
+ try:
1361
+ await asyncio.wait_for(
1362
+ api_client.agent_dispatch.create_dispatch(
1363
+ api.CreateAgentDispatchRequest(
1364
+ agent_name=dispatch_agent_name,
1365
+ room=room_name,
1366
+ metadata=_dispatch_metadata_json(agent_definition),
1367
+ )
1368
+ ),
1369
+ timeout=connect_timeout,
1370
+ )
1371
+ except asyncio.TimeoutError:
1372
+ outcome = _failure_outcome(
1373
+ TestCaseStatus.TIMED_OUT,
1374
+ FailureStage.PREPARING,
1375
+ "livekit_dispatch_timeout",
1376
+ "Target agent dispatch exceeded its deadline",
1377
+ retryable=True,
1378
+ )
1379
+ return outcome
1380
+ except Exception as exc:
1381
+ logger.warning(
1382
+ "LiveKit target dispatch failed",
1383
+ exc_info=redacted_exc_info(exc),
1384
+ extra={
1385
+ "run_id": run_id,
1386
+ "test_case_id": test_case_id,
1387
+ "room_name": room_name,
1388
+ },
1389
+ )
1390
+ outcome = _failure_outcome(
1391
+ TestCaseStatus.FAILED,
1392
+ FailureStage.PREPARING,
1393
+ "livekit_dispatch_failed",
1394
+ "Failed to dispatch the target agent",
1395
+ details=_safe_provider_error_details(
1396
+ exc, operation="agent_dispatch"
1397
+ ),
1398
+ )
1399
+ return outcome
1400
+ logger.info(
1401
+ "livekit_target_dispatched agent=%s room=%s run=%s case=%s",
1402
+ dispatch_agent_name,
1403
+ room_name,
1404
+ run_id,
1405
+ test_case_id,
1406
+ )
1407
+ if profile.uses_web_audio_bridge:
1408
+ try:
1409
+ connector = profile.build_connector(
1410
+ provider_target,
1411
+ conversation_direction=conversation_direction,
1412
+ )
1413
+ audio_bridge = LiveKitAudioBridge(
1414
+ url=str(runtime.url),
1415
+ api_key=api_key,
1416
+ api_secret=api_secret,
1417
+ room_name=room_name,
1418
+ identity=bridge_identity or "fagi-provider-bridge",
1419
+ connector=connector,
1420
+ )
1421
+ await asyncio.wait_for(
1422
+ audio_bridge.connect(), timeout=connect_timeout
1423
+ )
1424
+ provider_call_id = audio_bridge.call_id
1425
+ bridge_task = asyncio.create_task(audio_bridge.run())
1426
+ except asyncio.TimeoutError:
1427
+ outcome = _failure_outcome(
1428
+ TestCaseStatus.TIMED_OUT,
1429
+ FailureStage.PREPARING,
1430
+ "web_bridge_start_timeout",
1431
+ "Provider web call creation exceeded its deadline",
1432
+ retryable=True,
1433
+ )
1434
+ return outcome
1435
+ except Exception as exc:
1436
+ logger.warning(
1437
+ "Provider web bridge creation failed",
1438
+ exc_info=redacted_exc_info(exc),
1439
+ extra={
1440
+ "run_id": run_id,
1441
+ "test_case_id": test_case_id,
1442
+ "transport": transport.kind,
1443
+ },
1444
+ )
1445
+ outcome = _failure_outcome(
1446
+ TestCaseStatus.FAILED,
1447
+ FailureStage.PREPARING,
1448
+ "web_bridge_start_failed",
1449
+ "Failed to start the provider web bridge",
1450
+ details=_safe_provider_error_details(
1451
+ exc, operation="web_bridge_start"
1452
+ ),
1453
+ )
1454
+ return outcome
1455
+ if profile.places_outbound_call and api_client is not None:
1456
+ try:
1457
+ logger.info(
1458
+ "sip_outbound_dialing",
1459
+ extra={
1460
+ "run_id": run_id,
1461
+ "test_case_id": test_case_id,
1462
+ "room_name": room_name,
1463
+ },
1464
+ )
1465
+ await asyncio.wait_for(
1466
+ api_client.sip.create_sip_participant(
1467
+ api.CreateSIPParticipantRequest(
1468
+ sip_trunk_id=transport.sip_trunk_id,
1469
+ sip_number=transport.sip_number,
1470
+ sip_call_to=transport.sip_call_to,
1471
+ room_name=room_name,
1472
+ participant_identity=sip_participant_identity,
1473
+ wait_until_answered=True,
1474
+ play_ringtone=True,
1475
+ )
1476
+ ),
1477
+ timeout=sip_answer_timeout,
1478
+ )
1479
+ except asyncio.TimeoutError:
1480
+ outcome = _failure_outcome(
1481
+ TestCaseStatus.TIMED_OUT,
1482
+ FailureStage.PREPARING,
1483
+ "sip_answer_timeout",
1484
+ "Outbound SIP call was not answered before the deadline",
1485
+ retryable=True,
1486
+ )
1487
+ return outcome
1488
+ except Exception as exc:
1489
+ logger.warning(
1490
+ "SIP dial failed",
1491
+ exc_info=redacted_exc_info(exc),
1492
+ extra={
1493
+ "run_id": run_id,
1494
+ "test_case_id": test_case_id,
1495
+ "room_name": room_name,
1496
+ },
1497
+ )
1498
+ outcome = _failure_outcome(
1499
+ TestCaseStatus.FAILED,
1500
+ FailureStage.PREPARING,
1501
+ "sip_dial_failed",
1502
+ "Failed to dial the SIP participant",
1503
+ details=_safe_provider_error_details(exc, operation="sip_dial"),
1504
+ )
1505
+ return outcome
1506
+ if profile.receives_inbound_call:
1507
+ logger.info(
1508
+ "sip_inbound_ready",
1509
+ extra={
1510
+ "run_id": run_id,
1511
+ "test_case_id": test_case_id,
1512
+ "room_name": room_name,
1513
+ "sip_dispatch_rule_id": sip_dispatch_rule_id,
1514
+ "sip_dispatch_rule_created": sip_dispatch_rule_created,
1515
+ },
1516
+ )
1517
+ if runtime.room_name_verbatim and profile.receives_inbound_call:
1518
+ unexpected = _unexpected_participants(
1519
+ room,
1520
+ simulator_identity=simulator_identity,
1521
+ recorder_identity=recorder_identity,
1522
+ )
1523
+ if unexpected:
1524
+ logger.warning(
1525
+ "leased room occupied before dial",
1526
+ extra={
1527
+ "run_id": run_id,
1528
+ "test_case_id": test_case_id,
1529
+ "room_name": room_name,
1530
+ "unexpected_participants": len(unexpected),
1531
+ },
1532
+ )
1533
+ outcome = _failure_outcome(
1534
+ TestCaseStatus.FAILED,
1535
+ FailureStage.PREPARING,
1536
+ "livekit_room_occupied",
1537
+ "Another participant was already in the leased simulator room before the call was placed",
1538
+ retryable=True,
1539
+ )
1540
+ return outcome
1541
+ if transport.inbound_call_originator is not None:
1542
+ name = transport.inbound_call_originator
1543
+ try:
1544
+ call_originator = build_call_originator(transport)
1545
+ originated_call = await asyncio.wait_for(
1546
+ call_originator.start(), timeout=connect_timeout
1547
+ )
1548
+ provider_call_id = originated_call.call_id
1549
+ except asyncio.TimeoutError:
1550
+ outcome = _failure_outcome(
1551
+ TestCaseStatus.TIMED_OUT,
1552
+ FailureStage.PREPARING,
1553
+ f"{name}_call_start_timeout",
1554
+ f"{name.capitalize()} call creation exceeded its deadline",
1555
+ retryable=True,
1556
+ )
1557
+ return outcome
1558
+ except Exception as exc:
1559
+ logger.warning(
1560
+ f"{name.capitalize()} call creation failed",
1561
+ exc_info=redacted_exc_info(exc),
1562
+ extra={
1563
+ "run_id": run_id,
1564
+ "test_case_id": test_case_id,
1565
+ },
1566
+ )
1567
+ outcome = _failure_outcome(
1568
+ TestCaseStatus.FAILED,
1569
+ FailureStage.PREPARING,
1570
+ f"{name}_call_start_failed",
1571
+ f"Failed to start the {name.capitalize()} call",
1572
+ details=_safe_provider_error_details(
1573
+ exc, operation=f"{name}_call_start"
1574
+ ),
1575
+ )
1576
+ return outcome
1577
+ target = await _wait_for_target_audio(
1578
+ room,
1579
+ excluded_identities={simulator_identity, recorder_identity},
1580
+ target_identity=effective_target_identity,
1581
+ timeout=effective_readiness_timeout,
1582
+ )
1583
+ if (
1584
+ runtime.room_name_verbatim
1585
+ and profile.receives_inbound_call
1586
+ and transport.originator_from_number
1587
+ ):
1588
+ verdict = _caller_matches(
1589
+ target.attributes, transport.originator_from_number
1590
+ )
1591
+ if verdict is False:
1592
+ logger.warning(
1593
+ "leased room wrong caller",
1594
+ extra={
1595
+ "run_id": run_id,
1596
+ "test_case_id": test_case_id,
1597
+ "expected_digits": len(
1598
+ _number_digits(transport.originator_from_number)
1599
+ ),
1600
+ "observed_digits": len(
1601
+ _number_digits(
1602
+ target.attributes.get(
1603
+ _SIP_REMOTE_NUMBER_ATTRIBUTE, ""
1604
+ )
1605
+ )
1606
+ ),
1607
+ },
1608
+ )
1609
+ caller_verification = "mismatch"
1610
+ raise _LeasedRoomCallerMismatch()
1611
+ elif verdict is None:
1612
+ logger.warning(
1613
+ "leased room caller unverified",
1614
+ extra={"run_id": run_id, "test_case_id": test_case_id},
1615
+ )
1616
+ caller_verification = "unverified"
1617
+ else:
1618
+ caller_verification = "matched"
1619
+ logger.info(
1620
+ "livekit_target_joined identity=%s sid=%s track=%s run=%s case=%s",
1621
+ target.identity,
1622
+ target.sid,
1623
+ target.audio_track_sid,
1624
+ run_id,
1625
+ test_case_id,
1626
+ )
1627
+ # RoomIO auto-links to the first participant that joined — the
1628
+ # recorder, which publishes no audio — so the simulator's STT never
1629
+ # hears the target. Re-point it at the target readiness selected.
1630
+ target_room_io = getattr(session, "room_io", None)
1631
+ if target_room_io is not None:
1632
+ target_room_io.set_participant(target.identity)
1633
+
1634
+ # Swap the early buffer handler for the authoritative one. The
1635
+ # unregister → def → register sequence has NO await between the
1636
+ # unregister and register, so no header can land unhandled in the gap
1637
+ # (LiveKit allows one handler per topic and raises on double-register).
1638
+ if target_transcription_handler_registered:
1639
+ room.unregister_text_stream_handler(TOPIC_TRANSCRIPTION)
1640
+ target_transcription_handler_registered = False
1641
+
1642
+ # The target's mic track stays open for the whole call, so STT never
1643
+ # finalizes a turn; the agent's authoritative turns arrive on its
1644
+ # lk.transcription stream instead. Consume that stream and feed each
1645
+ # completed target utterance into the simulator as a user turn.
1646
+ def on_target_transcription(
1647
+ reader: "rtc.TextStreamReader",
1648
+ participant_identity: str,
1649
+ ) -> None:
1650
+ nonlocal target_transcription_mode
1651
+ attrs = reader.info.attributes or {}
1652
+ transcribed_track_id = attrs.get(ATTRIBUTE_TRANSCRIPTION_TRACK_ID)
1653
+ if transcribed_track_id:
1654
+ if transcribed_track_id != target.audio_track_sid:
1655
+ return
1656
+ elif str(participant_identity) != target.identity:
1657
+ return
1658
+ # First target transcription means the target is speaking — stop
1659
+ # the redundant simulator STT so it cannot emit duplicate turns.
1660
+ if not target_transcription_mode:
1661
+ session.input.set_audio_enabled(False)
1662
+ session.clear_user_turn()
1663
+ target_transcription_mode = True
1664
+ task = asyncio.create_task(
1665
+ _forward_target_transcription(
1666
+ reader,
1667
+ session,
1668
+ conversation_ended=conversation_ended,
1669
+ captured_target_turns=captured_target_turns,
1670
+ )
1671
+ )
1672
+ target_transcription_tasks.add(task)
1673
+ task.add_done_callback(target_transcription_tasks.discard)
1674
+
1675
+ room.register_text_stream_handler(
1676
+ TOPIC_TRANSCRIPTION, on_target_transcription
1677
+ )
1678
+ target_transcription_handler_registered = True
1679
+
1680
+ # Drain greeting streams buffered before readiness through the
1681
+ # authoritative handler (it re-applies the strict track/identity
1682
+ # filter, so over-buffered non-target streams are dropped here).
1683
+ buffered_streams = list(pending_target_transcriptions)
1684
+ pending_target_transcriptions.clear()
1685
+ for buffered_reader, buffered_identity in buffered_streams:
1686
+ on_target_transcription(buffered_reader, buffered_identity)
1687
+
1688
+ opener: asyncio.Task[None] | None = None
1689
+ if conversation_direction == "simulator_first" or _answered_by_voicemail():
1690
+ # A mailbox speaks first and needs no watchdog to break a mutual silence.
1691
+ customer_agent.open_conversation()
1692
+ else:
1693
+ # The agent placed this call and should speak first. If it does not, the person
1694
+ # answers rather than both sides waiting for each other.
1695
+ opener = asyncio.create_task(
1696
+ _open_if_nobody_speaks_first(
1697
+ session,
1698
+ customer_agent,
1699
+ timeout_seconds=_OPEN_INSTEAD_AFTER_SECONDS,
1700
+ )
1701
+ )
1702
+ try:
1703
+ stop_reason = await _wait_for_conversation_end(
1704
+ room,
1705
+ session,
1706
+ customer_agent=customer_agent,
1707
+ target_identity=target.identity,
1708
+ timeout=max_seconds,
1709
+ conversation_direction=conversation_direction,
1710
+ agent_first_silence_timeout_seconds=agent_first_silence_timeout_seconds,
1711
+ provider_task=bridge_task,
1712
+ )
1713
+ finally:
1714
+ # However the conversation ended, including badly, the watchdog goes with it: a
1715
+ # pending task at loop close is noise in the log of every call.
1716
+ if opener is not None and not opener.done():
1717
+ opener.cancel()
1718
+ logger.info(
1719
+ "livekit_conversation_ended stop_reason=%s run=%s case=%s",
1720
+ stop_reason,
1721
+ run_id,
1722
+ test_case_id,
1723
+ )
1724
+ # End the call cleanly. First let the party that just spoke commit its
1725
+ # own final turn — a LiveKit turn only lands in history once its TTS
1726
+ # finishes — bounded so we do not wait on the other side. We do NOT
1727
+ # wait for the target's trailing speech: once the conversation has
1728
+ # ended, the target talking on is monologuing into a call the other
1729
+ # side left.
1730
+ conversation_ended.set()
1731
+ if stop_reason == "simulator_end_call":
1732
+ wait_for_end_speech = getattr(
1733
+ customer_agent,
1734
+ "wait_for_end_speech",
1735
+ None,
1736
+ )
1737
+ try:
1738
+ if callable(wait_for_end_speech):
1739
+ await asyncio.wait_for(
1740
+ wait_for_end_speech(),
1741
+ timeout=_FINAL_TURN_COMMIT_WAIT_SECONDS,
1742
+ )
1743
+ except asyncio.TimeoutError:
1744
+ logger.warning(
1745
+ "Simulator closing speech did not finish before cleanup",
1746
+ extra={"run_id": run_id, "test_case_id": test_case_id},
1747
+ )
1748
+ _loop = asyncio.get_running_loop()
1749
+ _commit_deadline = _loop.time() + _FINAL_TURN_COMMIT_WAIT_SECONDS
1750
+ while _loop.time() < _commit_deadline:
1751
+ try:
1752
+ if session.current_speech is None:
1753
+ break
1754
+ except Exception: # noqa: BLE001
1755
+ break
1756
+ await asyncio.sleep(0.2)
1757
+ # Delete the room so the target agent can't keep monologuing into a
1758
+ # dead call (its audio would be recorded but is untranscribable once
1759
+ # the simulator has left) — the recording then ends when the call
1760
+ # actually ends, matching the transcript.
1761
+ if api_client is not None and managed_room_owned:
1762
+ try:
1763
+ await asyncio.wait_for(
1764
+ api_client.room.delete_room(
1765
+ api.DeleteRoomRequest(room=room_name)
1766
+ ),
1767
+ timeout=_cleanup_budget(),
1768
+ )
1769
+ except Exception as exc: # noqa: BLE001
1770
+ logger.warning(
1771
+ "LiveKit room delete on conversation end failed",
1772
+ exc_info=redacted_exc_info(exc),
1773
+ )
1774
+ messages = _canonical_report_messages(session)
1775
+ messages = _merge_captured_target_turns(messages, captured_target_turns)
1776
+ outcome = _conversation_outcome(
1777
+ stop_reason,
1778
+ messages,
1779
+ min_turn_messages=min_turn_messages,
1780
+ )
1781
+ except asyncio.TimeoutError:
1782
+ stage = (
1783
+ FailureStage.READINESS
1784
+ if session is not None and target is None
1785
+ else FailureStage.PREPARING
1786
+ )
1787
+ if stage == FailureStage.READINESS and profile.receives_inbound_call:
1788
+ code = "sip_inbound_no_participant"
1789
+ message = "No inbound SIP participant joined before deadline"
1790
+ elif stage == FailureStage.READINESS:
1791
+ code = "agent_unavailable"
1792
+ message = "Target agent did not become ready"
1793
+ else:
1794
+ code = "livekit_connect_timeout"
1795
+ message = "LiveKit setup exceeded its deadline"
1796
+ status = (
1797
+ TestCaseStatus.AGENT_UNAVAILABLE
1798
+ if stage == FailureStage.READINESS
1799
+ else TestCaseStatus.TIMED_OUT
1800
+ )
1801
+ outcome = _failure_outcome(
1802
+ status,
1803
+ stage,
1804
+ code,
1805
+ message,
1806
+ retryable=True,
1807
+ )
1808
+ except _LeasedRoomCallerMismatch:
1809
+ # Unlike the pre-dial failures above, this path has a billed call and
1810
+ # a joined target; the recordings, provider evidence and metadata
1811
+ # update below all run after the try, so this must not return early.
1812
+ outcome = _failure_outcome(
1813
+ TestCaseStatus.FAILED,
1814
+ FailureStage.READINESS,
1815
+ "livekit_room_wrong_caller",
1816
+ "The participant that answered is not calling from the customer's configured number",
1817
+ retryable=True,
1818
+ )
1819
+ except Exception as exc:
1820
+ logger.error(
1821
+ "LiveKit test case failed",
1822
+ exc_info=redacted_exc_info(exc),
1823
+ extra={
1824
+ "run_id": run_id,
1825
+ "test_case_id": test_case_id,
1826
+ "exception_type": type(exc).__name__,
1827
+ },
1828
+ )
1829
+ outcome = _failure_outcome(
1830
+ TestCaseStatus.FAILED,
1831
+ FailureStage.RUNNING if session is not None else FailureStage.PREPARING,
1832
+ "livekit_case_failed",
1833
+ "LiveKit test case failed",
1834
+ details={"exception_type": type(exc).__name__},
1835
+ )
1836
+ finally:
1837
+ # The ambience belongs to the caller agent, not the engine. Guarded because teardown
1838
+ # must never be the reason a case fails.
1839
+ if customer_agent is not None:
1840
+ try:
1841
+ # Bounded like every other teardown step. Closing the ambience player unpublishes
1842
+ # its track, and when the room's signal client has already died, that wait never
1843
+ # returns: the SDK loops on resume and restart while this await sits here, and the
1844
+ # case never completes, so the whole run is discarded on its deadline with a
1845
+ # finished conversation inside it. Measured: two calls ended on endCall at 15 and
1846
+ # 17 messages and both reported 570004ms and no test case.
1847
+ await asyncio.wait_for(
1848
+ customer_agent._stop_background_audio(),
1849
+ timeout=_cleanup_budget(
1850
+ _BACKGROUND_AUDIO_CLEANUP_TIMEOUT_SECONDS
1851
+ ),
1852
+ )
1853
+ except Exception:
1854
+ logger.warning("background audio not closed cleanly", exc_info=True)
1855
+ if target_transcription_handler_registered:
1856
+ room.unregister_text_stream_handler(TOPIC_TRANSCRIPTION)
1857
+ pending_target_transcriptions.clear()
1858
+ pending_transcriptions = list(target_transcription_tasks)
1859
+ for pending in pending_transcriptions:
1860
+ pending.cancel()
1861
+ if pending_transcriptions:
1862
+ # Cancelled above, but a task blocked reading a stream whose connection is gone does
1863
+ # not observe the cancellation, so this is bounded too.
1864
+ try:
1865
+ await asyncio.wait_for(
1866
+ asyncio.gather(*pending_transcriptions, return_exceptions=True),
1867
+ timeout=_cleanup_budget(),
1868
+ )
1869
+ except Exception as exc: # noqa: BLE001 - teardown never fails a case
1870
+ logger.warning("transcription tasks did not stop cleanly: %s", exc)
1871
+ session_to_close = session or (
1872
+ getattr(customer_agent, "started_session", None)
1873
+ if customer_agent is not None
1874
+ else None
1875
+ )
1876
+ if session_to_close is not None:
1877
+ try:
1878
+ await _close_agent_session(
1879
+ session_to_close,
1880
+ timeout=_cleanup_budget(_SESSION_CLEANUP_TIMEOUT_SECONDS),
1881
+ )
1882
+ except Exception as exc:
1883
+ _record_cleanup_error(
1884
+ cleanup_errors,
1885
+ exc,
1886
+ "session_close",
1887
+ run_id,
1888
+ test_case_id,
1889
+ )
1890
+ if models is not None:
1891
+ try:
1892
+ await asyncio.wait_for(
1893
+ models.aclose(),
1894
+ timeout=_cleanup_budget(),
1895
+ )
1896
+ except Exception as exc:
1897
+ _record_cleanup_error(
1898
+ cleanup_errors,
1899
+ exc,
1900
+ "models_close",
1901
+ run_id,
1902
+ test_case_id,
1903
+ )
1904
+ if recorder is not None:
1905
+ try:
1906
+ await asyncio.wait_for(
1907
+ recorder.aclose(),
1908
+ timeout=_cleanup_budget(),
1909
+ )
1910
+ except Exception as exc:
1911
+ _record_cleanup_error(
1912
+ cleanup_errors,
1913
+ exc,
1914
+ "recorder_close",
1915
+ run_id,
1916
+ test_case_id,
1917
+ )
1918
+ if audio_bridge is not None:
1919
+ try:
1920
+ await asyncio.wait_for(
1921
+ audio_bridge.aclose(), timeout=_cleanup_budget()
1922
+ )
1923
+ if bridge_task is not None:
1924
+ await asyncio.wait_for(bridge_task, timeout=_cleanup_budget())
1925
+ except Exception as exc:
1926
+ _record_cleanup_error(
1927
+ cleanup_errors,
1928
+ exc,
1929
+ "bridge_close",
1930
+ run_id,
1931
+ test_case_id,
1932
+ )
1933
+ if room_connected:
1934
+ try:
1935
+ await asyncio.wait_for(room.disconnect(), timeout=_cleanup_budget())
1936
+ except Exception as exc:
1937
+ _record_cleanup_error(
1938
+ cleanup_errors,
1939
+ exc,
1940
+ "room_disconnect",
1941
+ run_id,
1942
+ test_case_id,
1943
+ )
1944
+ if call_originator is not None:
1945
+ # Guarded like every other cleanup step below: a future escape
1946
+ # from the helper (today it never raises) must not skip the
1947
+ # dispatch-rule/room/api-client teardown that follows.
1948
+ try:
1949
+ finalize_result = await finalize_originator(
1950
+ call_originator,
1951
+ provider_call_id=provider_call_id,
1952
+ originator_name=transport.inbound_call_originator,
1953
+ case_started_at=case_started_at,
1954
+ cleanup_timeout=_cleanup_budget(),
1955
+ )
1956
+ for operation, exc in finalize_result.cleanup_errors:
1957
+ _record_cleanup_error(
1958
+ cleanup_errors, exc, operation, run_id, test_case_id
1959
+ )
1960
+ if finalize_result.termination_source:
1961
+ provider_termination_source = finalize_result.termination_source
1962
+ reconciled_call_ids = finalize_result.reconciled_call_ids
1963
+ if provider_call_id is None and reconciled_call_ids:
1964
+ # We stopped a billed call we believe is ours, so
1965
+ # evidence fetches its record instead of reporting
1966
+ # "not matched" after searching nothing.
1967
+ provider_call_id = reconciled_call_ids[0]
1968
+ # Written here too (not only in the metadata.update below)
1969
+ # because a start-failure path returns from inside the try,
1970
+ # bypassing that block entirely.
1971
+ if outcome is not None:
1972
+ outcome.metadata["reconciled_call_ids"] = reconciled_call_ids
1973
+ if finalize_result.termination_source:
1974
+ outcome.metadata["provider_termination_source"] = (
1975
+ finalize_result.termination_source
1976
+ )
1977
+ except Exception as exc:
1978
+ _record_cleanup_error(
1979
+ cleanup_errors,
1980
+ exc,
1981
+ f"{transport.inbound_call_originator}_call_finalize",
1982
+ run_id,
1983
+ test_case_id,
1984
+ )
1985
+ if (
1986
+ api_client is not None
1987
+ and sip_dispatch_rule_id
1988
+ and sip_dispatch_rule_created
1989
+ ):
1990
+ try:
1991
+ await asyncio.wait_for(
1992
+ _delete_sip_dispatch_rule(api_client, sip_dispatch_rule_id),
1993
+ timeout=_cleanup_budget(),
1994
+ )
1995
+ except Exception as exc:
1996
+ if not _is_not_found(exc):
1997
+ _record_cleanup_error(
1998
+ cleanup_errors,
1999
+ exc,
2000
+ "sip_dispatch_delete",
2001
+ run_id,
2002
+ test_case_id,
2003
+ )
2004
+ if api_client is not None and managed_room_owned:
2005
+ try:
2006
+ await asyncio.wait_for(
2007
+ api_client.room.delete_room(
2008
+ api.DeleteRoomRequest(room=room_name)
2009
+ ),
2010
+ timeout=_cleanup_budget(),
2011
+ )
2012
+ except Exception as exc:
2013
+ if not _is_not_found(exc):
2014
+ _record_cleanup_error(
2015
+ cleanup_errors,
2016
+ exc,
2017
+ "room_delete",
2018
+ run_id,
2019
+ test_case_id,
2020
+ )
2021
+ if api_client is not None:
2022
+ try:
2023
+ await api_client.aclose()
2024
+ except Exception as exc:
2025
+ _record_cleanup_error(
2026
+ cleanup_errors,
2027
+ exc,
2028
+ "api_close",
2029
+ run_id,
2030
+ test_case_id,
2031
+ )
2032
+ # Written here too (not only in the metadata.update below) because a
2033
+ # pre-dial return (e.g. the occupancy check) exits from inside the
2034
+ # try, bypassing that block entirely.
2035
+ if outcome is not None and outcome.metadata.get("cleanup_status") is None:
2036
+ outcome.metadata["cleanup_status"] = (
2037
+ "failed" if cleanup_errors else "completed"
2038
+ )
2039
+ outcome.metadata["cleanup_errors"] = cleanup_errors
2040
+ if outcome is None:
2041
+ outcome = _failure_outcome(
2042
+ TestCaseStatus.FAILED,
2043
+ FailureStage.FINALIZING,
2044
+ "livekit_outcome_missing",
2045
+ "LiveKit test case ended without an outcome",
2046
+ )
2047
+ if recorder is not None:
2048
+ _attach_recordings(
2049
+ outcome,
2050
+ recorder,
2051
+ simulator_identity=simulator_identity,
2052
+ target_identity=target.identity if target is not None else None,
2053
+ target_track_sid=(
2054
+ target.audio_track_sid if target is not None else None
2055
+ ),
2056
+ case_directory=case_directory,
2057
+ sample_rate=recorder_sample_rate,
2058
+ )
2059
+ if recorder.errors:
2060
+ cleanup_errors.extend(
2061
+ f"recording:{type(error).__name__}" for error in recorder.errors
2062
+ )
2063
+ if agent_definition.provider_evidence is not None:
2064
+ provider_summary, provider_artifacts = await _collect_provider_evidence(
2065
+ config=agent_definition.provider_evidence,
2066
+ transport=transport,
2067
+ run_id=run_id,
2068
+ test_case_id=test_case_id,
2069
+ case_directory=case_directory,
2070
+ started_at=case_started_at,
2071
+ target=target,
2072
+ provider_call_id_hint=provider_call_id,
2073
+ provider_api_key=_target_api_key(provider_target),
2074
+ provider_api_base_url=_target_evidence_base_url(provider_target),
2075
+ termination_source=provider_termination_source,
2076
+ )
2077
+ if provider_summary is not None:
2078
+ outcome.evidence.append(provider_summary)
2079
+ if provider_call_id is None:
2080
+ resolved_call_id = provider_summary.metadata.get("call_id")
2081
+ if resolved_call_id:
2082
+ provider_call_id = str(resolved_call_id)
2083
+ _reconcile_provider_observation(outcome, provider_summary)
2084
+ _recover_successful_provider_end_call(outcome, provider_summary)
2085
+ outcome.provider_artifacts.extend(provider_artifacts)
2086
+ outcome.metadata.update(
2087
+ {
2088
+ "simulator_participant_identity": simulator_identity,
2089
+ "target_participant_identity": (
2090
+ target.identity if target is not None else None
2091
+ ),
2092
+ "target_participant_sid": target.sid if target is not None else None,
2093
+ "target_audio_track_sid": (
2094
+ target.audio_track_sid if target is not None else None
2095
+ ),
2096
+ "target_participant_attributes": (
2097
+ dict(target.attributes) if target is not None else {}
2098
+ ),
2099
+ "cleanup_status": "failed" if cleanup_errors else "completed",
2100
+ "cleanup_errors": cleanup_errors,
2101
+ "sip_dispatch_rule_id": sip_dispatch_rule_id,
2102
+ "sip_dispatch_rule_created": sip_dispatch_rule_created,
2103
+ "target_provider": (
2104
+ provider_target.provider if provider_target is not None else None
2105
+ ),
2106
+ "provider_call_id": provider_call_id,
2107
+ "reconciled_call_ids": reconciled_call_ids,
2108
+ "caller_verification": caller_verification,
2109
+ "vapi_call_id": (
2110
+ provider_call_id
2111
+ if profile.evidence_provider == "vapi"
2112
+ or transport.inbound_call_originator == "vapi"
2113
+ else None
2114
+ ),
2115
+ "retell_call_id": (
2116
+ provider_call_id
2117
+ if profile.evidence_provider == "retell"
2118
+ or transport.inbound_call_originator == "retell"
2119
+ else None
2120
+ ),
2121
+ "simulator_model_usage": (
2122
+ customer_agent.model_usage
2123
+ if customer_agent is not None
2124
+ and hasattr(customer_agent, "model_usage")
2125
+ else []
2126
+ ),
2127
+ }
2128
+ )
2129
+ logger.info(
2130
+ "livekit_case_outcome status=%s stop_reason=%s failure=%s run=%s case=%s",
2131
+ outcome.status.value,
2132
+ outcome.metadata.get("stop_reason"),
2133
+ outcome.failure.code if outcome.failure is not None else None,
2134
+ run_id,
2135
+ test_case_id,
2136
+ )
2137
+ return outcome
2138
+
2139
+ async def _create_customer_agent(
2140
+ self,
2141
+ persona: Persona,
2142
+ simulator: SimulatorAgentDefinition | None,
2143
+ *,
2144
+ call_type: CallType = "inbound",
2145
+ agent_name: str | None = None,
2146
+ min_turn_messages: int = 0,
2147
+ ) -> tuple[_TestRunnerAgent, LiveKitModels]:
2148
+ customer_prompt = build_voice_simulator_prompt(
2149
+ persona,
2150
+ call_type=call_type,
2151
+ agent_name=agent_name,
2152
+ additional_instructions=(
2153
+ simulator.instructions if simulator is not None else None
2154
+ ),
2155
+ default_language=(
2156
+ simulator.stt.language if simulator is not None else None
2157
+ ),
2158
+ variables={"instruction": persona.situation or ""},
2159
+ # Delivery cues are Cartesia only. Passing the provider here rather than reading it
2160
+ # inside the prompt keeps the decision where the provider is actually known.
2161
+ tts_provider=(simulator.tts.provider if simulator is not None else None),
2162
+ )
2163
+ if simulator is None:
2164
+ voice_provider = os.environ.get(
2165
+ "SIMULATOR_VOICE_PROVIDER", "openai"
2166
+ ).lower()
2167
+ llm_config = _default_simulator_llm_config()
2168
+ stt_config = STTConfig(
2169
+ provider=voice_provider,
2170
+ model=os.environ.get("SIMULATOR_STT_MODEL", "gpt-4o-mini-transcribe"),
2171
+ )
2172
+ tts_config = TTSConfig(
2173
+ provider=voice_provider,
2174
+ model=os.environ.get("SIMULATOR_TTS_MODEL", "gpt-4o-mini-tts"),
2175
+ voice=os.environ.get("SIMULATOR_TTS_VOICE_ID", "alloy"),
2176
+ )
2177
+ instructions = customer_prompt
2178
+ allow_interruptions = None
2179
+ min_endpointing_delay = None
2180
+ max_endpointing_delay = None
2181
+ use_aligned_transcript = None
2182
+ else:
2183
+ llm_config = simulator.llm
2184
+ stt_config = simulator.stt
2185
+ tts_config = simulator.tts
2186
+ instructions = customer_prompt
2187
+ allow_interruptions = simulator.allow_interruptions
2188
+ min_endpointing_delay = simulator.min_endpointing_delay
2189
+ max_endpointing_delay = simulator.max_endpointing_delay
2190
+ use_aligned_transcript = simulator.use_tts_aligned_transcript
2191
+ # Per-persona voice: a persona may carry a ``voice`` (or ``voice_id``)
2192
+ # attribute so different simulated customers sound different. Deepgram
2193
+ # encodes the voice in the model name (``aura-*``); every other provider
2194
+ # uses the dedicated ``voice`` field. Falls back to the simulator/global
2195
+ # default when the persona does not specify one.
2196
+ persona_attrs = getattr(persona, "persona", None)
2197
+ persona_voice = (
2198
+ (persona_attrs.get("voice") or persona_attrs.get("voice_id"))
2199
+ if isinstance(persona_attrs, dict)
2200
+ else None
2201
+ )
2202
+ if persona_voice:
2203
+ field = "model" if tts_config.provider == "deepgram" else "voice"
2204
+ tts_config = tts_config.model_copy(update={field: str(persona_voice)})
2205
+ models = await build_livekit_models(
2206
+ llm_config=llm_config,
2207
+ stt_config=stt_config,
2208
+ tts_config=tts_config,
2209
+ )
2210
+ vad = await asyncio.to_thread(_load_silero_vad_sync)
2211
+ self._last_simulator_setup = {
2212
+ "instructions": instructions,
2213
+ "llm_config": llm_config,
2214
+ "stt_config": stt_config,
2215
+ "tts_config": tts_config,
2216
+ "allow_interruptions": allow_interruptions,
2217
+ "min_endpointing_delay": min_endpointing_delay,
2218
+ "max_endpointing_delay": max_endpointing_delay,
2219
+ "use_tts_aligned_transcript": use_aligned_transcript,
2220
+ }
2221
+ agent = _TestRunnerAgent(
2222
+ persona=persona,
2223
+ min_turn_messages=min_turn_messages,
2224
+ stt=models.stt,
2225
+ llm=models.llm,
2226
+ tts=models.tts,
2227
+ vad=vad,
2228
+ instructions=instructions,
2229
+ turn_handling=_simulator_turn_handling(
2230
+ vad=vad,
2231
+ allow_interruptions=allow_interruptions,
2232
+ min_endpointing_delay=min_endpointing_delay,
2233
+ max_endpointing_delay=max_endpointing_delay,
2234
+ ),
2235
+ use_tts_aligned_transcript=use_aligned_transcript,
2236
+ )
2237
+ return agent, models
2238
+
2239
+
2240
+ async def _wait_for_target_audio(
2241
+ room: rtc.Room,
2242
+ *,
2243
+ excluded_identities: set[str],
2244
+ target_identity: str | None,
2245
+ timeout: float,
2246
+ ) -> _TargetParticipant:
2247
+ ready = asyncio.Event()
2248
+ selected: _TargetParticipant | None = None
2249
+
2250
+ def inspect_room(*_args) -> None:
2251
+ nonlocal selected
2252
+ selected = _find_target_audio(
2253
+ room,
2254
+ excluded_identities=excluded_identities,
2255
+ target_identity=target_identity,
2256
+ )
2257
+ if selected is not None:
2258
+ ready.set()
2259
+
2260
+ room.on("participant_connected", inspect_room)
2261
+ room.on("track_published", inspect_room)
2262
+ room.on("track_subscribed", inspect_room)
2263
+ inspect_room()
2264
+ try:
2265
+ await asyncio.wait_for(ready.wait(), timeout=timeout)
2266
+ finally:
2267
+ _remove_room_listener(room, "participant_connected", inspect_room)
2268
+ _remove_room_listener(room, "track_published", inspect_room)
2269
+ _remove_room_listener(room, "track_subscribed", inspect_room)
2270
+ if selected is None:
2271
+ raise asyncio.TimeoutError
2272
+ return selected
2273
+
2274
+
2275
+ async def _forward_target_transcription(
2276
+ reader: "rtc.TextStreamReader",
2277
+ session: "AgentSession",
2278
+ *,
2279
+ conversation_ended: "asyncio.Event | None" = None,
2280
+ captured_target_turns: list[dict[str, Any]] | None = None,
2281
+ ) -> None:
2282
+ # Receiver-side wall clock — same clock domain as the simulator's
2283
+ # ChatMessage.metrics, and the target's transcript IO is playback-synced
2284
+ # (TranscriptSynchronizer), so stream-open ~= speech start and read_all()
2285
+ # completion ~= speech end. Timestamps embedded in the stream are the
2286
+ # sender's (laptop) clock; skew there would corrupt the derived latencies.
2287
+ started_at = time.time()
2288
+ try:
2289
+ transcript = (await reader.read_all()).strip()
2290
+ stopped_at = time.time()
2291
+ if not transcript:
2292
+ return
2293
+ # Capture the target's turn independently of the simulator session FIRST.
2294
+ # Once the session drains it rejects new input ("speech scheduling is
2295
+ # paused"), so a closing delivered after the simulator is done never
2296
+ # reaches the chat context. This list is merged into the report so the
2297
+ # trailing target turn survives regardless of session state.
2298
+ if captured_target_turns is not None:
2299
+ captured_target_turns.append(
2300
+ {
2301
+ "content": transcript,
2302
+ "started_speaking_at": started_at,
2303
+ "stopped_speaking_at": stopped_at,
2304
+ }
2305
+ )
2306
+ # Only elicit a simulator response while the conversation is live; once
2307
+ # it has ended the target's turn is recorded but the simulator stays
2308
+ # silent. The turn MUST travel through ``generate_reply(user_input=...)``:
2309
+ # the reply pipeline reads the agent's own chat context, not
2310
+ # ``session.history``, so a turn only added to the history is invisible
2311
+ # to the simulator LLM (it answers as if it heard nothing). The pipeline
2312
+ # then persists the message into both contexts once the reply schedules.
2313
+ if conversation_ended is None or not conversation_ended.is_set():
2314
+ try:
2315
+ session.generate_reply(user_input=transcript)
2316
+ except RuntimeError:
2317
+ # Session is already closing; the turn is captured above.
2318
+ pass
2319
+ else:
2320
+ return
2321
+ # Conversation over (or the session rejected the reply): record the
2322
+ # turn on the transcript without eliciting a response.
2323
+ try:
2324
+ session.history.add_message(role="user", content=transcript)
2325
+ except Exception: # noqa: BLE001
2326
+ pass
2327
+ except Exception as exc: # noqa: BLE001
2328
+ logger.warning(
2329
+ "Failed to consume target transcription stream",
2330
+ exc_info=redacted_exc_info(exc),
2331
+ )
2332
+
2333
+
2334
+ def _find_target_audio(
2335
+ room: rtc.Room,
2336
+ *,
2337
+ excluded_identities: set[str],
2338
+ target_identity: str | None,
2339
+ ) -> _TargetParticipant | None:
2340
+ candidates: list[tuple[int, int, _TargetParticipant]] = []
2341
+ agent_kind = getattr(
2342
+ rtc.ParticipantKind,
2343
+ "PARTICIPANT_KIND_AGENT",
2344
+ None,
2345
+ )
2346
+ for participant in room.remote_participants.values():
2347
+ identity = str(participant.identity)
2348
+ if identity in excluded_identities:
2349
+ continue
2350
+ if target_identity is not None and identity != target_identity:
2351
+ continue
2352
+ priority = 0 if getattr(participant, "kind", None) == agent_kind else 1
2353
+ for publication in participant.track_publications.values():
2354
+ if getattr(publication, "kind", None) != rtc.TrackKind.KIND_AUDIO:
2355
+ continue
2356
+ # Agents may publish ambient music/noise alongside their synthesized speech. The
2357
+ # first LiveKit publication is not necessarily the conversational track (the
2358
+ # official drive-thru example publishes ``background_audio``). Prefer ordinary
2359
+ # speech/microphone tracks so STT, transcription filtering, and recording all bind
2360
+ # to the same semantic stream.
2361
+ track_name = str(getattr(publication, "name", "") or "").lower()
2362
+ background = any(
2363
+ marker in track_name
2364
+ for marker in ("background", "ambient", "music", "sound_effect")
2365
+ )
2366
+ track_priority = 1 if background else 0
2367
+ attrs = dict(getattr(participant, "attributes", {}) or {})
2368
+ candidates.append(
2369
+ (
2370
+ priority,
2371
+ track_priority,
2372
+ _TargetParticipant(
2373
+ identity=identity,
2374
+ sid=str(participant.sid),
2375
+ audio_track_sid=str(publication.sid),
2376
+ attributes={
2377
+ str(key): str(value) for key, value in attrs.items()
2378
+ },
2379
+ ),
2380
+ )
2381
+ )
2382
+ if not candidates:
2383
+ return None
2384
+ return sorted(
2385
+ candidates,
2386
+ key=lambda item: (
2387
+ item[0],
2388
+ item[1],
2389
+ item[2].identity,
2390
+ item[2].audio_track_sid,
2391
+ ),
2392
+ )[0][2]
2393
+
2394
+
2395
+ # A run ends naturally when the simulator calls ``endCall``; this is only the
2396
+ # backstop for a conversation that has genuinely stalled or already finished but
2397
+ # never hung up. Kept long so normal turn-gaps (STT endpoint + LLM + TTS latency)
2398
+ # never trip it — the run is never cut off at a message count.
2399
+ _SILENCE_BACKSTOP_SECONDS = 90.0
2400
+
2401
+ # Mutual silence in a conversation both sides joined is a finished call, not a stalled one. A
2402
+ # thinking agent is working rather than silent, so the timer holds while either side is busy: no
2403
+ # fixed window fits both a 4.3s and a 25.2s reply. The measured fallback covers providers that
2404
+ # report no thinking state, stretching to the slowest reply this call has seen.
2405
+ _SETTLED_SILENCE_FLOOR_SECONDS = 12.0
2406
+ _SETTLED_LATENCY_MULTIPLE = 2.0
2407
+
2408
+ # LiveKit reports OUR SIMULATED CALLER as "assistant" and the TARGET AGENT as "user", because the
2409
+ # caller is this session's agent and the target connects as the remote party. The published
2410
+ # transcript swaps them (see _canonical_report_messages), so session-native code must never reuse
2411
+ # the published convention. Named here because reading it the wrong way round is silent: a check
2412
+ # still runs, still passes its tests, and watches the wrong side of the call.
2413
+ _CALLER = "assistant"
2414
+ _TARGET = "user"
2415
+
2416
+ # AgentState describes THIS SESSION'S AGENT, our caller; UserState describes the target. UserState
2417
+ # has no "thinking", so this pair stops us cutting off our own caller mid-thought and cannot see a
2418
+ # target composing a reply. The measured window below is what protects a slow target.
2419
+ _AGENT_BUSY_STATES = frozenset({"initializing", "thinking", "speaking"})
2420
+ _USER_BUSY_STATES = frozenset({"speaking"})
2421
+
2422
+
2423
+ def _either_side_busy(session: Any) -> bool:
2424
+ """Whether work is in flight, as opposed to a conversation that has gone quiet."""
2425
+ return (
2426
+ getattr(session, "agent_state", None) in _AGENT_BUSY_STATES
2427
+ or getattr(session, "user_state", None) in _USER_BUSY_STATES
2428
+ )
2429
+
2430
+
2431
+ def _settled_silence_window(
2432
+ observed_agent_reply: float, backstop_seconds: float
2433
+ ) -> float:
2434
+ """How long silence must last before a settled call is treated as over."""
2435
+ return min(
2436
+ backstop_seconds,
2437
+ max(
2438
+ _SETTLED_SILENCE_FLOOR_SECONDS,
2439
+ observed_agent_reply * _SETTLED_LATENCY_MULTIPLE,
2440
+ ),
2441
+ )
2442
+
2443
+
2444
+ def _turn_gap_seconds(
2445
+ previous: dict[str, Any], current: dict[str, Any]
2446
+ ) -> float | None:
2447
+ """Silence between one turn finishing and the next starting, in seconds.
2448
+
2449
+ Uses the real audio timing the transport reports and falls back to the wall-clock stamp for
2450
+ text-only turns. Returns None when neither side is timed, so a caller can tell "no gap" from
2451
+ "not measurable" rather than reading an absent measurement as zero.
2452
+ """
2453
+ start = current.get("started_speaking_at") or current.get("created_at") or None
2454
+ end = previous.get("stopped_speaking_at") or previous.get("created_at") or None
2455
+ if not start or not end:
2456
+ return None
2457
+ gap = float(start) - float(end)
2458
+ return gap if gap >= 0 else None
2459
+
2460
+
2461
+ def _observed_agent_reply_seconds(messages: list[dict[str, Any]]) -> float:
2462
+ """The slowest reply this agent has actually produced on this call.
2463
+
2464
+ Read from the transport rather than tracked against the poll loop's own clock, so it is also
2465
+ correct for history that arrives in bulk (a resume, a reconnect) where there was no live
2466
+ transition to observe. Prefers LiveKit's reported end-to-end latency and falls back to the gap
2467
+ between the caller finishing and the agent starting, which is the wait a listener would hear.
2468
+ """
2469
+ slowest = 0.0
2470
+ previous: dict[str, Any] | None = None
2471
+ for message in messages:
2472
+ if not message.get("content"):
2473
+ continue
2474
+ if message.get("role") == _TARGET:
2475
+ reported = message.get("e2e_latency")
2476
+ if reported:
2477
+ slowest = max(slowest, float(reported))
2478
+ elif previous is not None and previous.get("role") == _CALLER:
2479
+ gap = _turn_gap_seconds(previous, message)
2480
+ if gap is not None:
2481
+ slowest = max(slowest, gap)
2482
+ previous = message
2483
+ return slowest
2484
+
2485
+
2486
+ async def _wait_for_conversation_end(
2487
+ room: rtc.Room,
2488
+ session: AgentSession,
2489
+ *,
2490
+ customer_agent: _TestRunnerAgent,
2491
+ target_identity: str,
2492
+ timeout: float,
2493
+ conversation_direction: str,
2494
+ agent_first_silence_timeout_seconds: float,
2495
+ provider_task: asyncio.Task[None] | None = None,
2496
+ ) -> str:
2497
+ closed = asyncio.Event()
2498
+ target_disconnected = asyncio.Event()
2499
+ room_disconnected = asyncio.Event()
2500
+
2501
+ def on_close(_event) -> None:
2502
+ closed.set()
2503
+
2504
+ def on_participant_disconnected(participant) -> None:
2505
+ if str(participant.identity) == target_identity:
2506
+ target_disconnected.set()
2507
+
2508
+ def on_room_disconnected(*_args) -> None:
2509
+ room_disconnected.set()
2510
+
2511
+ session.on("close", on_close)
2512
+ room.on("participant_disconnected", on_participant_disconnected)
2513
+ # A native target commonly hangs up by DELETING the room (the LiveKit
2514
+ # hangup recipe); the simulator then sees a room disconnect, not a
2515
+ # participant_disconnected, and without this watcher the case idled
2516
+ # through the silence backstop before ending.
2517
+ room.on("disconnected", on_room_disconnected)
2518
+ # The target may have left in the gap between readiness and this
2519
+ # registration — the event is gone, so recheck presence once.
2520
+ remote_participants = getattr(room, "remote_participants", None)
2521
+ if isinstance(remote_participants, dict) and not any(
2522
+ str(participant.identity) == target_identity
2523
+ for participant in remote_participants.values()
2524
+ ):
2525
+ target_disconnected.set()
2526
+ tasks = {
2527
+ "closed": asyncio.create_task(closed.wait()),
2528
+ "target_disconnected": asyncio.create_task(target_disconnected.wait()),
2529
+ "room_disconnected": asyncio.create_task(room_disconnected.wait()),
2530
+ "simulator_end_call": asyncio.create_task(customer_agent.end_requested.wait()),
2531
+ "conversation_stalled": asyncio.create_task(
2532
+ _wait_for_conversation_silence(
2533
+ session,
2534
+ # A stub agent in a test carries no floor; absent means never settle early.
2535
+ min_turn_messages=int(
2536
+ getattr(customer_agent, "_min_turn_messages", 0) or 0
2537
+ ),
2538
+ )
2539
+ ),
2540
+ "closing_loop": asyncio.create_task(_wait_for_closing_loop(session)),
2541
+ "no_conversation": asyncio.create_task(
2542
+ _wait_for_conversation_never_started(
2543
+ session,
2544
+ timeout_seconds=_NO_CONVERSATION_TIMEOUT_SECONDS,
2545
+ )
2546
+ ),
2547
+ }
2548
+ if conversation_direction == "agent_first":
2549
+ tasks["conversation_silence_timeout"] = asyncio.create_task(
2550
+ _wait_for_agent_first_silence(
2551
+ session,
2552
+ timeout_seconds=agent_first_silence_timeout_seconds,
2553
+ )
2554
+ )
2555
+ if provider_task is not None:
2556
+ tasks["provider_disconnected"] = provider_task
2557
+ try:
2558
+ done, pending = await asyncio.wait(
2559
+ set(tasks.values()),
2560
+ timeout=timeout,
2561
+ return_when=asyncio.FIRST_COMPLETED,
2562
+ )
2563
+ owned_pending = {
2564
+ task
2565
+ for name, task in tasks.items()
2566
+ if task in pending and name != "provider_disconnected"
2567
+ }
2568
+ for task in owned_pending:
2569
+ task.cancel()
2570
+ if owned_pending:
2571
+ await asyncio.gather(*owned_pending, return_exceptions=True)
2572
+ if not done:
2573
+ return "timeout"
2574
+ # A crashed monitor is also "done"; it must not count as its condition.
2575
+ completed: set[str] = set()
2576
+ monitor_failures: dict[str, BaseException] = {}
2577
+ for name, task in tasks.items():
2578
+ if task not in done or task.cancelled():
2579
+ continue
2580
+ exc = task.exception()
2581
+ if exc is None:
2582
+ completed.add(name)
2583
+ else:
2584
+ monitor_failures[name] = exc
2585
+ for name, exc in monitor_failures.items():
2586
+ logger.warning(
2587
+ "conversation end monitor failed",
2588
+ exc_info=redacted_exc_info(exc),
2589
+ extra={"monitor": name, "target_identity": target_identity},
2590
+ )
2591
+ # A bridge task error is still a real provider-side disconnect.
2592
+ if "provider_disconnected" in monitor_failures:
2593
+ completed.add("provider_disconnected")
2594
+ for reason in (
2595
+ "simulator_end_call",
2596
+ "target_disconnected",
2597
+ "room_disconnected",
2598
+ "no_conversation",
2599
+ # A farewell loop is a finished conversation, so it outranks the silence backstop
2600
+ # that would otherwise report the same call as a stall.
2601
+ "closing_loop",
2602
+ "conversation_silence_timeout",
2603
+ "conversation_stalled",
2604
+ "provider_disconnected",
2605
+ "closed",
2606
+ ):
2607
+ if reason in completed:
2608
+ return "session_closed" if reason == "closed" else reason
2609
+ if monitor_failures:
2610
+ return "monitor_failed"
2611
+ return "session_closed"
2612
+ finally:
2613
+ _remove_room_listener(session, "close", on_close)
2614
+ _remove_room_listener(
2615
+ room,
2616
+ "participant_disconnected",
2617
+ on_participant_disconnected,
2618
+ )
2619
+ _remove_room_listener(room, "disconnected", on_room_disconnected)
2620
+
2621
+
2622
+ _CLOSING_PHRASES = (
2623
+ "goodbye",
2624
+ "bye",
2625
+ "take care",
2626
+ "have a great day",
2627
+ "have a good day",
2628
+ "have a wonderful day",
2629
+ "you too",
2630
+ )
2631
+
2632
+ _CLOSING_EXCHANGE_LIMIT = 4
2633
+
2634
+
2635
+ # Words a farewell is allowed to be made of. Anything outside this set is substance, whatever the
2636
+ # turn's length: "yes it is, bye" is an answer and ending on it would cut a live call short.
2637
+ _CLOSING_FILLER = frozenset(
2638
+ """
2639
+ a again alright and bye byebye care cheers day drive evening fine good goodbye great
2640
+ have later lovely morning much nice night ok okay perfect right safe see so soon sounds
2641
+ speak sure take talk thank thanks then to tomorrow too well wonderful you your
2642
+ """.split()
2643
+ )
2644
+
2645
+
2646
+ def _is_closing_only(text: str) -> bool:
2647
+ """Whether a turn is nothing but a farewell.
2648
+
2649
+ Deliberately narrow: a turn that closes AND carries anything else (a question, a fact, a
2650
+ correction) is still conversation, and ending on it would cut a live call short.
2651
+
2652
+ Decided on whether every word is farewell filler rather than on a word count. A cap of six
2653
+ words classified "Sounds great, thanks. Talk tomorrow. Bye." as a farewell and "Sounds good,
2654
+ talk to you then. Bye." as conversation, purely because the second has one more word, and the
2655
+ engine then asked the caller for two further turns and got two more goodbyes.
2656
+ """
2657
+ stripped = "".join(
2658
+ character.lower() if character.isalnum() or character.isspace() else " "
2659
+ for character in (text or "")
2660
+ ).split()
2661
+ if not stripped or len(stripped) > 12:
2662
+ return False
2663
+ joined = " ".join(stripped)
2664
+ if not any(phrase in joined for phrase in _CLOSING_PHRASES):
2665
+ return False
2666
+ return not (set(stripped) - _CLOSING_FILLER)
2667
+
2668
+
2669
+ def _stop_any_further_speech(session: Any) -> None:
2670
+ """Cancel anything already in flight, so the farewell is the last thing said.
2671
+
2672
+ Noticing the farewell only stops us asking for the NEXT turn. A reply already being generated
2673
+ still plays, which is how "Take care." arrived after a correct goodbye on a measured call. The
2674
+ farewell itself is already in history, meaning its own audio finished, so there is nothing of
2675
+ the caller's left to cut off here.
2676
+ """
2677
+ try:
2678
+ session.interrupt(force=True)
2679
+ except Exception: # noqa: BLE001 - nothing in flight, or a session already shutting down
2680
+ logger.debug("nothing to interrupt when the call was closed", exc_info=True)
2681
+
2682
+
2683
+ async def _wait_for_closing_loop(
2684
+ session: AgentSession,
2685
+ *,
2686
+ limit: int = _CLOSING_EXCHANGE_LIMIT,
2687
+ ) -> None:
2688
+ """Finish once both sides are only trading farewells.
2689
+
2690
+ A simulator that does not reach for ``endCall`` leaves the target answering goodbye with
2691
+ goodbye until the deadline. One such call ran seventy-six turns, held its worker past the
2692
+ world pool's patience and cost the rest of that job its worlds, so this ends the call on the
2693
+ evidence already in the transcript rather than waiting for a timeout that arrives too late.
2694
+ """
2695
+ while True:
2696
+ messages = _session_messages(session)
2697
+ spoken = [
2698
+ message for message in messages if (message.get("content") or "").strip()
2699
+ ]
2700
+ # The caller's own farewell is the end of the call from its side, so there is no reason to
2701
+ # ask it for another turn. Waiting for a loop of farewells is what produced "Talk
2702
+ # tomorrow. Bye." followed by "Take care." and then "Bye." -- three closings where the
2703
+ # first was already correct. Rule 10 of the caller's prompt says exactly this, and an
2704
+ # instruction cannot enforce it: the model only speaks again because it was asked to.
2705
+ if (
2706
+ _turns_from_each_side(spoken) >= 1
2707
+ and spoken
2708
+ and spoken[-1].get("role") == _CALLER
2709
+ and _is_closing_only(str(spoken[-1].get("content") or ""))
2710
+ ):
2711
+ logger.info("the caller said goodbye, ending the call")
2712
+ _stop_any_further_speech(session)
2713
+ return
2714
+ tail = spoken[-limit:]
2715
+ if len(tail) == limit and all(
2716
+ _is_closing_only(str(message.get("content") or "")) for message in tail
2717
+ ):
2718
+ logger.info(
2719
+ "closing loop: last %d turns were farewells only, ending the call",
2720
+ limit,
2721
+ )
2722
+ _stop_any_further_speech(session)
2723
+ return
2724
+ # A turn lands in history only after its TTS finishes, so every poll interval between the
2725
+ # farewell committing and this noticing is time in which the caller can be asked for
2726
+ # another turn. Measured: one trailing turn survived at a one-second poll.
2727
+ await asyncio.sleep(0.25)
2728
+
2729
+
2730
+ async def _wait_for_conversation_silence(
2731
+ session: AgentSession,
2732
+ *,
2733
+ quiet_seconds: float = _SILENCE_BACKSTOP_SECONDS,
2734
+ min_turn_messages: int = 0,
2735
+ ) -> None:
2736
+ """Finish only after a long, genuine stretch of mutual silence.
2737
+
2738
+ The simulator ends a call by calling ``endCall`` once the scenario is done;
2739
+ this is only the backstop for a conversation that has actually stalled (or
2740
+ already finished but never hung up). It deliberately does **not** look at the
2741
+ message count — a run is never cut off at a floor, it runs as long as turns
2742
+ keep flowing. The timer resets on every new message and while either side is
2743
+ speaking, so only a real ``quiet_seconds`` gap of nothing ends the call.
2744
+
2745
+ Parks until the first non-empty turn: a call where nobody ever spoke is the
2746
+ ``no_conversation`` monitor's condition, and this backstop firing first
2747
+ mislabeled dead calls as merely settled.
2748
+ """
2749
+ last_signature: tuple[tuple[str, str], ...] | None = None
2750
+ stable_since: float | None = None
2751
+ loop = asyncio.get_running_loop()
2752
+ while True:
2753
+ messages = _session_messages(session)
2754
+ signature = tuple((message["role"], message["content"]) for message in messages)
2755
+ if not any(message["content"] for message in messages):
2756
+ last_signature = signature
2757
+ stable_since = None
2758
+ await asyncio.sleep(0.1)
2759
+ continue
2760
+ now = loop.time()
2761
+ participant_busy = _either_side_busy(session)
2762
+ floor, _ = _turn_requirements(min_turn_messages)
2763
+ # Far enough in for the measured window to beat the fixed one. A third of the floor is a
2764
+ # threshold, not a derived figure: enough turns to have timed a reply, well short of done.
2765
+ settled = min_turn_messages > 0 and _turns_from_each_side(messages) >= max(
2766
+ 2, floor // 3
2767
+ )
2768
+ effective_quiet = (
2769
+ _settled_silence_window(
2770
+ _observed_agent_reply_seconds(messages), quiet_seconds
2771
+ )
2772
+ if settled
2773
+ else quiet_seconds
2774
+ )
2775
+ if participant_busy:
2776
+ stable_since = None
2777
+ elif stable_since is None or signature != last_signature:
2778
+ stable_since = now
2779
+ elif now - stable_since >= effective_quiet:
2780
+ return
2781
+ last_signature = signature
2782
+ await asyncio.sleep(0.1)
2783
+
2784
+
2785
+ async def _wait_for_conversation_never_started(
2786
+ session: AgentSession,
2787
+ *,
2788
+ timeout_seconds: float,
2789
+ ) -> None:
2790
+ """Completes only when no non-empty turn has ever been committed; parks
2791
+ forever (until cancelled) once the conversation has actually started."""
2792
+ loop = asyncio.get_running_loop()
2793
+ deadline = loop.time() + timeout_seconds
2794
+ while loop.time() < deadline:
2795
+ if any(message["content"] for message in _session_messages(session)):
2796
+ await asyncio.Event().wait()
2797
+ await asyncio.sleep(0.5)
2798
+
2799
+
2800
+ def _voicemail_tone_style() -> str:
2801
+ """The style whose tone this call plays; empty for a person, and for a full mailbox by design."""
2802
+ if not _answered_by_voicemail():
2803
+ return ""
2804
+ style = (
2805
+ os.environ.get("HARNESS_VOICEMAIL_STYLE", "").strip().lower()
2806
+ or _DEFAULT_VOICEMAIL_STYLE
2807
+ )
2808
+ return style if style in _VOICEMAIL_TONE_BY_STYLE else ""
2809
+
2810
+
2811
+ def _downloaded_audio(source: str) -> str | None:
2812
+ """A local copy of a remote audio file, or None: a call heard in the clear beats a dropped one."""
2813
+ import tempfile
2814
+ import urllib.request
2815
+
2816
+ try:
2817
+ suffix = ".mp3" if ".mp3" in source else ".ogg" if ".ogg" in source else ".wav"
2818
+ with urllib.request.urlopen(source, timeout=15) as response:
2819
+ data = response.read()
2820
+ handle = tempfile.NamedTemporaryFile(delete=False, suffix=suffix)
2821
+ handle.write(data)
2822
+ handle.close()
2823
+ return handle.name
2824
+ except Exception:
2825
+ return None
2826
+
2827
+
2828
+ def _frame_at_mixer_rate(frame: "rtc.AudioFrame") -> "rtc.AudioFrame":
2829
+ """The same audio at the mixer's rate, which reinterprets rather than resamples what it is given."""
2830
+ if frame.sample_rate == _BACKGROUND_MIXER_RATE:
2831
+ return frame
2832
+ resampler = _MIXER_RESAMPLERS.get((frame.sample_rate, frame.num_channels))
2833
+ if resampler is None:
2834
+ resampler = PCMResampler(
2835
+ from_rate=frame.sample_rate,
2836
+ to_rate=_BACKGROUND_MIXER_RATE,
2837
+ channels=frame.num_channels,
2838
+ )
2839
+ _MIXER_RESAMPLERS[(frame.sample_rate, frame.num_channels)] = resampler
2840
+ converted = resampler.convert(bytes(frame.data))
2841
+ return rtc.AudioFrame(
2842
+ data=converted,
2843
+ sample_rate=_BACKGROUND_MIXER_RATE,
2844
+ num_channels=frame.num_channels,
2845
+ samples_per_channel=len(converted) // (2 * frame.num_channels),
2846
+ )
2847
+
2848
+
2849
+ def _tone_frame(hz: float, seconds: float) -> "rtc.AudioFrame":
2850
+ """One frame of sine, faded in and out: a burst at full amplitude clicks and a detector hears the click."""
2851
+ total = int(_BACKGROUND_MIXER_RATE * seconds)
2852
+ fade = max(1, int(_BACKGROUND_MIXER_RATE * 0.01))
2853
+ samples = array.array("h")
2854
+ for index in range(total):
2855
+ gain = min(1.0, index / fade, (total - index) / fade)
2856
+ samples.append(
2857
+ int(
2858
+ 32767
2859
+ * 0.9
2860
+ * gain
2861
+ * math.sin(2 * math.pi * hz * index / _BACKGROUND_MIXER_RATE)
2862
+ )
2863
+ )
2864
+ return rtc.AudioFrame(
2865
+ data=samples.tobytes(),
2866
+ sample_rate=_BACKGROUND_MIXER_RATE,
2867
+ num_channels=1,
2868
+ samples_per_channel=total,
2869
+ )
2870
+
2871
+
2872
+ def _answered_by_voicemail() -> bool:
2873
+ """Whether a mailbox answered rather than a person; set per scenario by the call runner."""
2874
+ return os.environ.get("HARNESS_ANSWERED_BY", "").strip().lower() == "voicemail"
2875
+
2876
+
2877
+ async def _open_if_nobody_speaks_first(
2878
+ session: AgentSession,
2879
+ customer_agent: Any,
2880
+ *,
2881
+ timeout_seconds: float,
2882
+ ) -> None:
2883
+ """Have the simulated person open the conversation when the other side never does.
2884
+
2885
+ Only for a call the agent was supposed to start. It opens exactly the way a simulator-first call
2886
+ does, through ``open_conversation``, so the person's own initial message is used where the
2887
+ persona has one. Returns as soon as anybody speaks, which is the ordinary case.
2888
+ """
2889
+ loop = asyncio.get_running_loop()
2890
+ deadline = loop.time() + timeout_seconds
2891
+ while loop.time() < deadline:
2892
+ if any(message["content"] for message in _session_messages(session)):
2893
+ return
2894
+ await asyncio.sleep(0.2)
2895
+ if any(message["content"] for message in _session_messages(session)):
2896
+ return
2897
+ logger.warning(
2898
+ "no first turn after %ss; the simulated person opens instead", timeout_seconds
2899
+ )
2900
+ try:
2901
+ customer_agent.open_conversation()
2902
+ except Exception: # noqa: BLE001 - a call that cannot be opened is the case's own failure
2903
+ logger.warning("the simulated person could not open the call", exc_info=True)
2904
+
2905
+
2906
+ async def _wait_for_agent_first_silence(
2907
+ session: AgentSession,
2908
+ *,
2909
+ timeout_seconds: float,
2910
+ ) -> None:
2911
+ last_signature: tuple[tuple[str, str], ...] = ()
2912
+ last_change = asyncio.get_running_loop().time()
2913
+ while True:
2914
+ messages = _session_messages(session)
2915
+ signature = tuple((message["role"], message["content"]) for message in messages)
2916
+ # A turn lands in history only after its TTS finishes, so an in-flight utterance longer
2917
+ # than the timeout must count as activity -- and so must an agent that is still thinking,
2918
+ # or the caller opens over the top of a reply that was on its way.
2919
+ if signature != last_signature or _either_side_busy(session):
2920
+ last_signature = signature
2921
+ last_change = asyncio.get_running_loop().time()
2922
+ roles = {message["role"] for message in messages if message["content"]}
2923
+ if {"user", "assistant"}.issubset(
2924
+ roles
2925
+ ) and asyncio.get_running_loop().time() - last_change >= timeout_seconds:
2926
+ return
2927
+ await asyncio.sleep(0.1)
2928
+
2929
+
2930
+ def _session_messages(session: AgentSession) -> list[dict[str, Any]]:
2931
+ """Return normalized transcript messages with real per-item speech timing.
2932
+
2933
+ Each dict carries:
2934
+ role, content: str
2935
+ started_speaking_at, stopped_speaking_at: float | None
2936
+ Real audio timing from ``ChatMessage.metrics`` (seconds since epoch).
2937
+ See livekit.agents.llm.chat_context.MetricsReport.
2938
+ created_at: float
2939
+ Fallback wall-clock stamp from ``ChatMessage.created_at`` (used when
2940
+ the metrics timestamps are missing, e.g. text-only turns).
2941
+ interrupted: bool
2942
+ e2e_latency: float | None
2943
+ Agent-side turn latency, when reported by LiveKit.
2944
+
2945
+ Downstream code turns these into millisecond offsets so the platform can
2946
+ recompute WPM, talk-ratio and interruption counts with real overlap data.
2947
+ """
2948
+ messages: list[dict[str, Any]] = []
2949
+ for item in session.history.items:
2950
+ if getattr(item, "type", None) != "message":
2951
+ continue
2952
+ role = getattr(item, "role", None)
2953
+ text = getattr(item, "text_content", None)
2954
+ if role is None or text is None:
2955
+ continue
2956
+ interrupted = bool(getattr(item, "interrupted", False))
2957
+ created_at = float(getattr(item, "created_at", 0.0) or 0.0)
2958
+ metrics = getattr(item, "metrics", None) or {}
2959
+ started_speaking_at = _maybe_float(metrics.get("started_speaking_at"))
2960
+ stopped_speaking_at = _maybe_float(metrics.get("stopped_speaking_at"))
2961
+ e2e_latency = _maybe_float(metrics.get("e2e_latency"))
2962
+ current: dict[str, Any] = {
2963
+ "role": str(role),
2964
+ "content": str(text),
2965
+ "created_at": created_at,
2966
+ "started_speaking_at": started_speaking_at,
2967
+ "stopped_speaking_at": stopped_speaking_at,
2968
+ "interrupted": interrupted,
2969
+ "e2e_latency": e2e_latency,
2970
+ }
2971
+ if messages and messages[-1]["role"] == current["role"]:
2972
+ previous = messages[-1]
2973
+ previous_text = previous["content"]
2974
+ if current["content"].startswith(previous_text):
2975
+ # Newer emission extends the previous partial — keep the
2976
+ # earliest start we saw, adopt the latest stop.
2977
+ current["started_speaking_at"] = (
2978
+ previous.get("started_speaking_at")
2979
+ or current["started_speaking_at"]
2980
+ )
2981
+ current["created_at"] = previous["created_at"] or created_at
2982
+ messages[-1] = current
2983
+ elif previous_text.startswith(current["content"]):
2984
+ previous["interrupted"] = previous.get("interrupted") or interrupted
2985
+ previous["stopped_speaking_at"] = (
2986
+ previous.get("stopped_speaking_at")
2987
+ or current["stopped_speaking_at"]
2988
+ )
2989
+ elif previous.get("interrupted") or interrupted:
2990
+ previous["content"] = f"{previous_text} {current['content']}".strip()
2991
+ previous["interrupted"] = interrupted
2992
+ previous["stopped_speaking_at"] = current[
2993
+ "stopped_speaking_at"
2994
+ ] or previous.get("stopped_speaking_at")
2995
+ else:
2996
+ messages.append(current)
2997
+ continue
2998
+ messages.append(current)
2999
+ return messages
3000
+
3001
+
3002
+ def _maybe_float(value: Any) -> float | None:
3003
+ if value is None:
3004
+ return None
3005
+ try:
3006
+ return float(value)
3007
+ except (TypeError, ValueError):
3008
+ return None
3009
+
3010
+
3011
+ def _canonical_report_messages(session: AgentSession) -> list[dict[str, Any]]:
3012
+ """Emit report messages with roles remapped to the test-agent perspective.
3013
+
3014
+ LiveKit reports our simulator as ``assistant`` and the target agent as
3015
+ ``user`` (the SDK connects with role ``agent``); we swap those so the
3016
+ downstream platform sees:
3017
+ role="user" → simulator / customer
3018
+ role="assistant" → agent-under-test
3019
+ which matches the CallTranscript convention.
3020
+
3021
+ Timing anchors (``started_speaking_at`` / ``stopped_speaking_at``) travel
3022
+ through unchanged so the platform can derive ms offsets.
3023
+ """
3024
+ role_map = {"assistant": "user", "user": "assistant"}
3025
+ messages: list[dict[str, Any]] = []
3026
+ for source in _session_messages(session):
3027
+ messages.append(
3028
+ {
3029
+ "role": role_map.get(source["role"], source["role"]),
3030
+ "content": source["content"],
3031
+ "created_at": source.get("created_at"),
3032
+ "started_speaking_at": source.get("started_speaking_at"),
3033
+ "stopped_speaking_at": source.get("stopped_speaking_at"),
3034
+ "interrupted": source.get("interrupted", False),
3035
+ "e2e_latency": source.get("e2e_latency"),
3036
+ }
3037
+ )
3038
+ return messages
3039
+
3040
+
3041
+ def _merge_captured_target_turns(
3042
+ messages: list[dict[str, Any]],
3043
+ captured_target_turns: list[dict[str, Any]] | None,
3044
+ ) -> list[dict[str, Any]]:
3045
+ """Restore native target-turn timing, and append any target turn that never
3046
+ reached the session history.
3047
+
3048
+ Native target turns are fed to the simulator via ``generate_reply(
3049
+ user_input=text)`` — a text input with no audio metrics — so their report
3050
+ entries carry a start but no ``stopped_speaking_at``: zero-duration turns
3051
+ that leave bot WPM, latency, and talk-ratio unpopulated. Each captured turn
3052
+ carries receiver-side wall-clock timing (see ``_forward_target_transcription``);
3053
+ here we (a) patch it onto the matching ``assistant`` turns missing a real
3054
+ stop, and (b) append the trailing turn delivered after the simulator drained
3055
+ ("speech scheduling is paused"). One turn can arrive as several partial or
3056
+ extended emissions, so match by containment and aggregate min-start/max-stop.
3057
+ Only populated by the native transcription handler — VAPI/Retell are untouched.
3058
+ """
3059
+ if not captured_target_turns:
3060
+ return messages
3061
+
3062
+ def _matching(text: str) -> list[dict[str, Any]]:
3063
+ result = []
3064
+ for captured in captured_target_turns:
3065
+ cap_text = (captured.get("content") or "").strip()
3066
+ if cap_text and (text in cap_text or cap_text in text):
3067
+ result.append(captured)
3068
+ return result
3069
+
3070
+ # (a) Fill timing onto existing assistant turns that lack a real stop; never
3071
+ # override genuine audio metrics if livekit-agents ever populates them.
3072
+ for message in messages:
3073
+ if message.get("role") != "assistant":
3074
+ continue
3075
+ text = (message.get("content") or "").strip()
3076
+ if not text:
3077
+ continue
3078
+ started = message.get("started_speaking_at")
3079
+ stopped = message.get("stopped_speaking_at")
3080
+ if (
3081
+ isinstance(started, (int, float))
3082
+ and isinstance(stopped, (int, float))
3083
+ and stopped > started
3084
+ ):
3085
+ continue
3086
+ matched = _matching(text)
3087
+ starts = [
3088
+ c["started_speaking_at"]
3089
+ for c in matched
3090
+ if isinstance(c.get("started_speaking_at"), (int, float))
3091
+ ]
3092
+ stops = [
3093
+ c["stopped_speaking_at"]
3094
+ for c in matched
3095
+ if isinstance(c.get("stopped_speaking_at"), (int, float))
3096
+ ]
3097
+ if starts:
3098
+ message["started_speaking_at"] = min(starts)
3099
+ if not isinstance(message.get("created_at"), (int, float)):
3100
+ message["created_at"] = min(starts)
3101
+ if stops:
3102
+ message["stopped_speaking_at"] = max(stops)
3103
+
3104
+ # (b) Append target turns that never reached the report at all.
3105
+ assistant_texts = [
3106
+ (m.get("content") or "").strip()
3107
+ for m in messages
3108
+ if m.get("role") == "assistant" and m.get("content")
3109
+ ]
3110
+
3111
+ def _already_present(text: str) -> bool:
3112
+ return any(text in existing or existing in text for existing in assistant_texts)
3113
+
3114
+ last_ts = 0.0
3115
+ for m in messages:
3116
+ for key in ("stopped_speaking_at", "started_speaking_at", "created_at"):
3117
+ value = m.get(key)
3118
+ if isinstance(value, (int, float)) and value > last_ts:
3119
+ last_ts = value
3120
+
3121
+ merged = list(messages)
3122
+ for offset, captured in enumerate(captured_target_turns, start=1):
3123
+ text = (captured.get("content") or "").strip()
3124
+ if not text or _already_present(text):
3125
+ continue
3126
+ started = captured.get("started_speaking_at")
3127
+ if not isinstance(started, (int, float)):
3128
+ started = (last_ts + offset) if last_ts else None
3129
+ stopped = captured.get("stopped_speaking_at")
3130
+ if not isinstance(stopped, (int, float)):
3131
+ stopped = started
3132
+ merged.append(
3133
+ {
3134
+ "role": "assistant",
3135
+ "content": text,
3136
+ "created_at": started,
3137
+ "started_speaking_at": started,
3138
+ "stopped_speaking_at": stopped,
3139
+ "interrupted": False,
3140
+ "e2e_latency": None,
3141
+ }
3142
+ )
3143
+ assistant_texts.append(text)
3144
+ return merged
3145
+
3146
+
3147
+ def _caller_never_spoke(messages: list[dict[str, Any]]) -> bool:
3148
+ """Whether the simulated caller's turns exist as text with no audio behind them.
3149
+
3150
+ Speech synthesis that fails still leaves the caller's line in the transcript, so a mute
3151
+ simulator and a silent agent produce the same stall unless the missing audio is read directly.
3152
+ """
3153
+
3154
+ def timed(message: dict[str, Any]) -> bool:
3155
+ return isinstance(message.get("started_speaking_at"), (int, float))
3156
+
3157
+ spoken = [
3158
+ message
3159
+ for message in messages
3160
+ if message.get("role") == "user" and (message.get("content") or "").strip()
3161
+ ]
3162
+ if not spoken or any(timed(message) for message in spoken):
3163
+ return False
3164
+ # Only the agent's turns carrying timing makes the caller's missing timing evidence of
3165
+ # silence rather than a transcript that simply does not record when anyone spoke.
3166
+ return any(
3167
+ timed(message) for message in messages if message.get("role") == "assistant"
3168
+ )
3169
+
3170
+
3171
+ def _recover_successful_provider_end_call(
3172
+ outcome: _CaseOutcome,
3173
+ provider_summary: EvidenceSourceSummary,
3174
+ ) -> None:
3175
+ """Keep a clean provider hangup separate from scenario correctness.
3176
+
3177
+ Retell can disconnect immediately after its ``end_call`` tool succeeds, before the final
3178
+ synthesized farewell is committed into LiveKit's transcript. A minimum-turn guard may have
3179
+ provisionally classified that as an incomplete conversation. Provider evidence is the
3180
+ authoritative lifecycle signal here: promote the call to completed and let scenario checks
3181
+ report any business-goal failure. Requiring both roles protects genuine mute/no-conversation
3182
+ failures from being hidden by a malformed provider trace.
3183
+ """
3184
+ if outcome.failure is None or outcome.failure.code not in {
3185
+ "insufficient_conversation",
3186
+ "target_disconnected",
3187
+ "room_disconnected",
3188
+ }:
3189
+ return
3190
+ if not _has_role_alternation(outcome.messages):
3191
+ return
3192
+ calls = provider_summary.metadata.get("tool_calls")
3193
+ if not isinstance(calls, list):
3194
+ return
3195
+ ended_cleanly = any(
3196
+ isinstance(call, dict)
3197
+ and str(call.get("name") or "").strip().lower() == "end_call"
3198
+ and call.get("ok") is not False
3199
+ for call in calls
3200
+ )
3201
+ if not ended_cleanly:
3202
+ return
3203
+ outcome.status = TestCaseStatus.COMPLETED
3204
+ outcome.failure = None
3205
+ outcome.metadata["provider_end_call_recovered"] = True
3206
+
3207
+
3208
+ def _reconcile_provider_observation(
3209
+ outcome: _CaseOutcome,
3210
+ provider_summary: EvidenceSourceSummary,
3211
+ ) -> None:
3212
+ """Recover provider-native speech and deterministic target tool failures."""
3213
+ raw_messages = provider_summary.metadata.get("messages")
3214
+ provider_messages: list[dict[str, Any]] = []
3215
+ if isinstance(raw_messages, list):
3216
+ for raw in raw_messages:
3217
+ if not isinstance(raw, dict):
3218
+ continue
3219
+ role = str(raw.get("role") or "").strip().lower()
3220
+ content = str(raw.get("content") or "").strip()
3221
+ if role not in {"user", "assistant"} or not content:
3222
+ continue
3223
+ provider_messages.append(dict(raw))
3224
+ if provider_messages and not outcome.messages:
3225
+ outcome.messages = provider_messages
3226
+ outcome.transcript = "\n".join(
3227
+ f"{message['role']}: {message['content']}" for message in provider_messages
3228
+ )
3229
+ outcome.metadata["provider_transcript_recovered"] = True
3230
+
3231
+ calls = provider_summary.metadata.get("tool_calls")
3232
+ failed_call = (
3233
+ next(
3234
+ (
3235
+ call
3236
+ for call in calls
3237
+ if isinstance(call, dict)
3238
+ and call.get("ok") is False
3239
+ and str(call.get("type") or "").lower() != "end_call"
3240
+ ),
3241
+ None,
3242
+ )
3243
+ if isinstance(calls, list)
3244
+ else None
3245
+ )
3246
+ if failed_call is None or outcome.failure is None:
3247
+ return
3248
+ if outcome.failure.code not in {
3249
+ "insufficient_conversation",
3250
+ "target_disconnected",
3251
+ "room_disconnected",
3252
+ "no_conversation",
3253
+ "conversation_stalled",
3254
+ "conversation_silence_timeout",
3255
+ }:
3256
+ return
3257
+ name = str(failed_call.get("name") or "unknown")
3258
+ error = str(failed_call.get("error") or "tool call failed")
3259
+ if len(error) > 500:
3260
+ error = error[:500]
3261
+ outcome.status = TestCaseStatus.FAILED
3262
+ outcome.failure = SimulationFailure(
3263
+ stage=FailureStage.RUNNING,
3264
+ code="target_agent_tool_failed",
3265
+ message=f"Target agent tool {name!r} failed: {error}",
3266
+ retryable=False,
3267
+ provider=str(provider_summary.metadata.get("provider") or "provider"),
3268
+ details={
3269
+ "tool_name": name,
3270
+ "provider_end_reason": str(
3271
+ provider_summary.metadata.get("end_reason") or ""
3272
+ ),
3273
+ },
3274
+ )
3275
+ outcome.metadata["provider_tool_failure_attributed"] = True
3276
+
3277
+
3278
+ def _turn_requirements(min_turn_messages: int) -> tuple[int, bool]:
3279
+ """The turn floor and whether alternation is required: a mailbox is held to its greeting alone."""
3280
+ if _answered_by_voicemail():
3281
+ return min(min_turn_messages, _VOICEMAIL_MIN_TURN_MESSAGES), False
3282
+ return min_turn_messages, True
3283
+
3284
+
3285
+ def _has_role_alternation(messages: list[dict[str, Any]]) -> bool:
3286
+ roles = {msg.get("role") for msg in messages if msg.get("content")}
3287
+ return "user" in roles and "assistant" in roles
3288
+
3289
+
3290
+ def _turns_from_each_side(messages: list[dict[str, Any]]) -> int:
3291
+ """How many turns the quieter speaker took; a total is inflated by one side's own filler."""
3292
+ spoken = [msg for msg in messages if msg.get("content")]
3293
+ return min(
3294
+ sum(1 for msg in spoken if msg.get("role") == "assistant"),
3295
+ sum(1 for msg in spoken if msg.get("role") == "user"),
3296
+ )
3297
+
3298
+
3299
+ # Two unanswered turns: one can be the caller finishing a thought, two means nobody is replying.
3300
+ _QUIET_AFTER_UNANSWERED_TURNS = 2
3301
+
3302
+
3303
+ def _target_has_gone_quiet(messages: list[dict[str, Any]]) -> bool:
3304
+ """Whether the agent has stopped replying, so the floor can never be reached honestly.
3305
+
3306
+ The floor counts messages, and the caller's own turns count toward it, so a caller that is
3307
+ refused the tool talks to fill the silence and eventually buys its own permission. That is the
3308
+ opposite of what the floor is for. When the agent has spoken and then stopped, the caller is
3309
+ allowed to hang up instead.
3310
+ """
3311
+ spoken = [message for message in messages if message.get("content")]
3312
+ if not any(message.get("role") == "assistant" for message in spoken):
3313
+ return False
3314
+ trailing = 0
3315
+ for message in reversed(spoken):
3316
+ if message.get("role") != "user":
3317
+ break
3318
+ trailing += 1
3319
+ return trailing >= _QUIET_AFTER_UNANSWERED_TURNS
3320
+
3321
+
3322
+ def _conversation_outcome(
3323
+ stop_reason: str,
3324
+ messages: list[dict[str, str]],
3325
+ *,
3326
+ min_turn_messages: int,
3327
+ ) -> _CaseOutcome:
3328
+ transcript = "\n".join(
3329
+ f"{message['role']}: {message['content']}" for message in messages
3330
+ )
3331
+ if (
3332
+ stop_reason in {"target_disconnected", "room_disconnected"}
3333
+ and _has_role_alternation(messages)
3334
+ and _has_natural_terminal_exchange(messages)
3335
+ ):
3336
+ # A provider target can deliberately end a short call before the generated
3337
+ # minimum-turn budget (for example, by accepting a caller's request to hang
3338
+ # up). The transport still completed successfully. Keep call lifecycle
3339
+ # separate from business-goal correctness: the scenario checks/evals decide
3340
+ # whether ending early was acceptable instead of reporting a false
3341
+ # connectivity failure.
3342
+ return _CaseOutcome(
3343
+ status=TestCaseStatus.COMPLETED,
3344
+ transcript=transcript,
3345
+ messages=messages,
3346
+ metadata={
3347
+ "stop_reason": stop_reason,
3348
+ "short_terminal_exchange": True,
3349
+ },
3350
+ )
3351
+ if (
3352
+ stop_reason == "conversation_silence_timeout"
3353
+ and len(messages) >= min_turn_messages
3354
+ and _has_role_alternation(messages)
3355
+ and _has_natural_terminal_exchange(messages)
3356
+ ):
3357
+ # Agent-first calls use a short silence watchdog because the tested
3358
+ # agent owns the opening turn. A simulator can occasionally omit its
3359
+ # endCall tool even after both sides have clearly closed the call. Do
3360
+ # not turn a fully recorded farewell/transfer into an infrastructure
3361
+ # failure merely because the now-idle room remained open. Evaluation
3362
+ # still decides whether the agent actually completed the requested
3363
+ # business action.
3364
+ return _CaseOutcome(
3365
+ status=TestCaseStatus.COMPLETED,
3366
+ transcript=transcript,
3367
+ messages=messages,
3368
+ metadata={
3369
+ "stop_reason": stop_reason,
3370
+ "terminal_exchange_recovered": True,
3371
+ },
3372
+ )
3373
+ if stop_reason == "timeout":
3374
+ return _failure_outcome(
3375
+ TestCaseStatus.TIMED_OUT,
3376
+ FailureStage.RUNNING,
3377
+ "conversation_timeout",
3378
+ "Conversation exceeded its deadline",
3379
+ transcript=transcript,
3380
+ messages=messages,
3381
+ retryable=True,
3382
+ )
3383
+ if stop_reason == "conversation_silence_timeout" and _caller_never_spoke(messages):
3384
+ # The target sat in real silence because nothing was ever spoken at it. Retrying cannot
3385
+ # put a voice back on the line, so fail fast and name the synthesis rather than spending
3386
+ # the attempt budget reporting the agent as stalled.
3387
+ return _failure_outcome(
3388
+ TestCaseStatus.FAILED,
3389
+ FailureStage.RUNNING,
3390
+ "simulator_tts_silent",
3391
+ "Simulated caller produced transcript text but no audio",
3392
+ transcript=transcript,
3393
+ messages=messages,
3394
+ retryable=False,
3395
+ details={"stop_reason": stop_reason, "turn_count": str(len(messages))},
3396
+ )
3397
+ stalled = {
3398
+ "conversation_silence_timeout",
3399
+ "conversation_stalled",
3400
+ "session_closed",
3401
+ "no_conversation",
3402
+ "monitor_failed",
3403
+ }
3404
+ if _answered_by_voicemail():
3405
+ # Silence after a mailbox greeting is the call's natural end, not a stall.
3406
+ stalled.discard("conversation_silence_timeout")
3407
+ if stop_reason in stalled:
3408
+ code = stop_reason
3409
+ message = {
3410
+ "conversation_silence_timeout": (
3411
+ "Agent-first conversation stalled after it began"
3412
+ ),
3413
+ "conversation_stalled": (
3414
+ "Conversation produced no new speech for the stall deadline"
3415
+ ),
3416
+ "session_closed": (
3417
+ "Conversation session closed before a natural end condition"
3418
+ ),
3419
+ "no_conversation": (
3420
+ "No conversation turns were committed before the inactivity deadline"
3421
+ ),
3422
+ "monitor_failed": (
3423
+ "Conversation end monitoring failed before a natural end condition"
3424
+ ),
3425
+ }[stop_reason]
3426
+ return _failure_outcome(
3427
+ TestCaseStatus.FAILED,
3428
+ FailureStage.RUNNING,
3429
+ code,
3430
+ message,
3431
+ transcript=transcript,
3432
+ messages=messages,
3433
+ retryable=True,
3434
+ )
3435
+ floor, alternation_required = _turn_requirements(min_turn_messages)
3436
+ if len(messages) < floor or (
3437
+ alternation_required and not _has_role_alternation(messages)
3438
+ ):
3439
+ code = (
3440
+ stop_reason
3441
+ if stop_reason in {"target_disconnected", "room_disconnected"}
3442
+ else "insufficient_conversation"
3443
+ )
3444
+ return _failure_outcome(
3445
+ TestCaseStatus.FAILED,
3446
+ FailureStage.RUNNING,
3447
+ code,
3448
+ "Conversation ended before the required alternating turns completed",
3449
+ transcript=transcript,
3450
+ messages=messages,
3451
+ retryable=stop_reason
3452
+ in {"target_disconnected", "room_disconnected", "session_closed"},
3453
+ details={
3454
+ "stop_reason": stop_reason,
3455
+ "turn_count": str(len(messages)),
3456
+ "minimum_turn_count": str(floor),
3457
+ },
3458
+ )
3459
+ return _CaseOutcome(
3460
+ status=TestCaseStatus.COMPLETED,
3461
+ transcript=transcript,
3462
+ messages=messages,
3463
+ metadata={"stop_reason": stop_reason},
3464
+ )
3465
+
3466
+
3467
+ def _has_natural_terminal_exchange(messages: list[dict[str, str]]) -> bool:
3468
+ """Recognize only explicit terminal language near the end of a call.
3469
+
3470
+ This deliberately avoids broad sentiment or short-answer heuristics. A
3471
+ normal unanswered question must remain a silence failure. The two safe
3472
+ cases are an explicit farewell, or a transfer handoff followed by the
3473
+ caller's acknowledgement.
3474
+ """
3475
+ tail = [
3476
+ (
3477
+ str(message.get("role") or "").lower(),
3478
+ str(message.get("content") or "").strip().lower(),
3479
+ )
3480
+ for message in messages[-4:]
3481
+ if str(message.get("content") or "").strip()
3482
+ ]
3483
+ if not tail:
3484
+ return False
3485
+ farewell_markers = (
3486
+ "goodbye",
3487
+ "bye",
3488
+ "take care",
3489
+ "have a great day",
3490
+ "have a good day",
3491
+ "have a nice day",
3492
+ )
3493
+ if any(marker in text for _role, text in tail for marker in farewell_markers):
3494
+ return True
3495
+
3496
+ for index, (role, text) in enumerate(tail[:-1]):
3497
+ if role != "assistant" or "transfer" not in text:
3498
+ continue
3499
+ if not any(marker in text for marker in ("now", "connect", "please wait")):
3500
+ continue
3501
+ next_role, acknowledgement = tail[index + 1]
3502
+ if next_role == "user" and acknowledgement.rstrip(".! ") in {
3503
+ "ok",
3504
+ "okay",
3505
+ "alright",
3506
+ "please do",
3507
+ "thank you",
3508
+ "thanks",
3509
+ }:
3510
+ return True
3511
+ return False
3512
+
3513
+
3514
+ def _failure_outcome(
3515
+ status: TestCaseStatus,
3516
+ stage: FailureStage,
3517
+ code: str,
3518
+ message: str,
3519
+ *,
3520
+ transcript: str = "",
3521
+ messages: list[dict[str, str]] | None = None,
3522
+ retryable: bool = False,
3523
+ details: dict[str, str] | None = None,
3524
+ ) -> _CaseOutcome:
3525
+ return _CaseOutcome(
3526
+ status=status,
3527
+ transcript=transcript,
3528
+ messages=messages or [],
3529
+ failure=SimulationFailure(
3530
+ stage=stage,
3531
+ code=code,
3532
+ message=message,
3533
+ retryable=retryable,
3534
+ provider="livekit",
3535
+ details=details or {},
3536
+ ),
3537
+ )
3538
+
3539
+
3540
+ def _record_simulator_setup(
3541
+ case_directory: Path,
3542
+ *,
3543
+ persona: Persona,
3544
+ instructions: str,
3545
+ llm_config: Any,
3546
+ stt_config: Any,
3547
+ tts_config: Any,
3548
+ turn_handling: Any = None,
3549
+ extra: dict[str, Any] | None = None,
3550
+ ) -> None:
3551
+ """Write the exact prompt and voice settings this call is about to use.
3552
+
3553
+ Reconstructing either one afterwards from a transcript is guesswork, and the simulator's
3554
+ prompt is what decides how the caller behaves. Written before the call connects so it
3555
+ survives a run that dies mid-conversation.
3556
+ """
3557
+
3558
+ def settings(config: Any) -> Any:
3559
+ if config is None:
3560
+ return None
3561
+ for method in ("model_dump", "dict"):
3562
+ dump = getattr(config, method, None)
3563
+ if callable(dump):
3564
+ try:
3565
+ return dump()
3566
+ except Exception: # noqa: BLE001 - never fail a call over logging
3567
+ pass
3568
+ return str(config)
3569
+
3570
+ try:
3571
+ case_directory.mkdir(parents=True, exist_ok=True)
3572
+ (case_directory / "simulator-prompt.txt").write_text(
3573
+ instructions or "", encoding="utf-8"
3574
+ )
3575
+ payload = {
3576
+ "persona": settings(persona),
3577
+ "simulator_system_prompt": instructions or "",
3578
+ "llm": settings(llm_config),
3579
+ "stt": settings(stt_config),
3580
+ "tts": settings(tts_config),
3581
+ "turn_handling": settings(turn_handling),
3582
+ }
3583
+ payload.update(extra or {})
3584
+ (case_directory / "simulator-setup.json").write_text(
3585
+ json.dumps(payload, indent=2, ensure_ascii=False, default=str),
3586
+ encoding="utf-8",
3587
+ )
3588
+ except Exception as error: # noqa: BLE001 - logging must never break a run
3589
+ logger.warning("could not record simulator setup: %s", error)
3590
+
3591
+
3592
+ def _attach_recordings(
3593
+ outcome: _CaseOutcome,
3594
+ recorder: RoomRecorder,
3595
+ *,
3596
+ simulator_identity: str,
3597
+ target_identity: str | None,
3598
+ target_track_sid: str | None,
3599
+ case_directory: Path,
3600
+ sample_rate: int,
3601
+ ) -> None:
3602
+ simulator_paths = recorder.paths_for_participant(simulator_identity)
3603
+ target_paths = (
3604
+ recorder.paths_for_participant(
3605
+ target_identity,
3606
+ track_sid=target_track_sid,
3607
+ )
3608
+ if target_identity is not None
3609
+ else []
3610
+ )
3611
+ # ``audio_track_sid`` identifies the track used to establish target
3612
+ # readiness, but it is not necessarily the track that remains published for
3613
+ # the conversation. Agents using ``BackgroundAudioPlayer`` publish more
3614
+ # than one audio track and can replace the initially selected publication.
3615
+ # Recording is evidence of the whole participant, so fall back to all of the
3616
+ # target participant's tracks instead of silently producing no artifact.
3617
+ if target_identity is not None and not target_paths:
3618
+ target_paths = recorder.paths_for_participant(target_identity)
3619
+
3620
+ # The simulator normally publishes with ``simulator_identity``. Retain a
3621
+ # conservative fallback for SDKs that expose the local publication under a
3622
+ # different participant identity: only use it when there is exactly one
3623
+ # non-target publishing participant, so another caller can never be folded
3624
+ # into the customer channel accidentally.
3625
+ if not simulator_paths:
3626
+ non_target_identities = {
3627
+ record.participant_identity
3628
+ for record in recorder.records
3629
+ if record.participant_identity != target_identity
3630
+ }
3631
+ if len(non_target_identities) == 1:
3632
+ simulator_paths = recorder.paths_for_participant(
3633
+ next(iter(non_target_identities))
3634
+ )
3635
+ audio_directory = case_directory / "audio"
3636
+ input_path = _collapse_recordings(
3637
+ simulator_paths,
3638
+ audio_directory / "simulator.wav",
3639
+ sample_rate=sample_rate,
3640
+ )
3641
+ output_path = _collapse_recordings(
3642
+ target_paths,
3643
+ audio_directory / "target.wav",
3644
+ sample_rate=sample_rate,
3645
+ )
3646
+ combined_path = mix_recordings(
3647
+ [path for path in (input_path, output_path) if path is not None],
3648
+ audio_directory / "combined.wav",
3649
+ sample_rate=sample_rate,
3650
+ )
3651
+ stereo_path = mix_recordings_stereo(
3652
+ [path for path in (input_path,) if path is not None],
3653
+ [path for path in (output_path,) if path is not None],
3654
+ audio_directory / "stereo.wav",
3655
+ sample_rate=sample_rate,
3656
+ )
3657
+ outcome.audio_input_path = str(input_path) if input_path is not None else None
3658
+ outcome.audio_output_path = str(output_path) if output_path is not None else None
3659
+ outcome.audio_combined_path = (
3660
+ str(combined_path) if combined_path is not None else None
3661
+ )
3662
+ outcome.audio_stereo_path = str(stereo_path) if stereo_path is not None else None
3663
+ outcome.metadata["recording_tracks"] = [
3664
+ {
3665
+ "participant_identity": record.participant_identity,
3666
+ "participant_sid": record.participant_sid,
3667
+ "track_sid": record.track_sid,
3668
+ "path": str(record.path),
3669
+ "start_offset_frames": record.start_offset_frames,
3670
+ }
3671
+ for record in recorder.records
3672
+ ]
3673
+ outcome.metadata["recording_diagnostics"] = {
3674
+ "simulator_identity": simulator_identity,
3675
+ "target_identity": target_identity,
3676
+ "target_track_sid": target_track_sid,
3677
+ "simulator_track_count": len(simulator_paths),
3678
+ "target_track_count": len(target_paths),
3679
+ "recorder_error_types": [type(error).__name__ for error in recorder.errors],
3680
+ }
3681
+ if combined_path is None:
3682
+ logger.warning(
3683
+ "LiveKit call completed without captured audio tracks",
3684
+ extra={
3685
+ "simulator_identity": simulator_identity,
3686
+ "target_identity": target_identity,
3687
+ "target_track_sid": target_track_sid,
3688
+ "recorded_track_count": len(recorder.records),
3689
+ "recorder_error_types": [
3690
+ type(error).__name__ for error in recorder.errors
3691
+ ],
3692
+ },
3693
+ )
3694
+ speech_starts = [
3695
+ float(message["started_speaking_at"])
3696
+ for message in outcome.messages
3697
+ if isinstance(message.get("started_speaking_at"), (int, float))
3698
+ ]
3699
+ if recorder.recording_started_at is not None and speech_starts:
3700
+ outcome.metadata["recording_offset_ms"] = max(
3701
+ 0,
3702
+ round((min(speech_starts) - recorder.recording_started_at) * 1000),
3703
+ )
3704
+
3705
+
3706
+ def _collapse_recordings(
3707
+ paths: list[Path],
3708
+ destination: Path,
3709
+ *,
3710
+ sample_rate: int,
3711
+ ) -> Path | None:
3712
+ if not paths:
3713
+ return None
3714
+ if len(paths) == 1:
3715
+ return paths[0]
3716
+ return mix_recordings(paths, destination, sample_rate=sample_rate)
3717
+
3718
+
3719
+ def _target_api_key(target: VoiceProviderTarget | None) -> str | None:
3720
+ if target is None:
3721
+ return None
3722
+ return os.environ.get(target.api_key_env) or None
3723
+
3724
+
3725
+ def _target_evidence_base_url(target: VoiceProviderTarget | None) -> str | None:
3726
+ if isinstance(target, VapiTargetConfig):
3727
+ return str(target.api_base_url).rstrip("/")
3728
+ if isinstance(target, RetellTargetConfig):
3729
+ parsed = urlsplit(str(target.api_url))
3730
+ return f"{parsed.scheme}://{parsed.netloc}"
3731
+ return None
3732
+
3733
+
3734
+ def _default_simulator_llm_config() -> LLMConfig:
3735
+ return LLMConfig(
3736
+ provider=os.environ.get("SIMULATOR_LLM_PROVIDER", "openai"),
3737
+ model=os.environ.get("SIMULATOR_LLM_MODEL", "gpt-4o-mini"),
3738
+ temperature=0.6,
3739
+ )
3740
+
3741
+
3742
+ def _resolve_livekit_runtime(
3743
+ agent_definition: AgentDefinition,
3744
+ runtime: LiveKitSimulatorRuntime | None,
3745
+ ) -> LiveKitSimulatorRuntime:
3746
+ if runtime is not None:
3747
+ return runtime
3748
+ if agent_definition.url is None or not agent_definition.room_name:
3749
+ raise ValueError(
3750
+ "livekit_runtime_required: provide LiveKitSimulatorRuntime or legacy "
3751
+ "AgentDefinition url and room_name"
3752
+ )
3753
+ return LiveKitSimulatorRuntime(
3754
+ url=agent_definition.url,
3755
+ room_name=agent_definition.room_name,
3756
+ room_mode=agent_definition.room_mode,
3757
+ )
3758
+
3759
+
3760
+ def _resolve_room_name(
3761
+ runtime: LiveKitSimulatorRuntime,
3762
+ *,
3763
+ run_id: str,
3764
+ test_case_id: str,
3765
+ index: int,
3766
+ invocation_id: str,
3767
+ ) -> str:
3768
+ rendered = runtime.room_name.format(
3769
+ run_id=run_id,
3770
+ test_case_id=test_case_id,
3771
+ index=index,
3772
+ invocation_id=invocation_id,
3773
+ )
3774
+ if runtime.room_mode == "external" or getattr(runtime, "room_name_verbatim", False):
3775
+ return rendered
3776
+ prefix = _SAFE_ROOM.sub("-", rendered).strip("-._") or "simulation"
3777
+ suffix_parts = []
3778
+ if invocation_id not in prefix:
3779
+ suffix_parts.append(invocation_id)
3780
+ if test_case_id not in prefix:
3781
+ suffix_parts.append(test_case_id[-12:])
3782
+ suffix = "-" + "-".join(suffix_parts) if suffix_parts else ""
3783
+ return f"{prefix[: 255 - len(suffix)]}{suffix}"
3784
+
3785
+
3786
+ def _has_room_template(room_name: str) -> bool:
3787
+ return any(
3788
+ marker in room_name for marker in ("{run_id}", "{test_case_id}", "{index}")
3789
+ )
3790
+
3791
+
3792
+ def _api_url(url: str) -> str:
3793
+ if url.startswith("wss://"):
3794
+ return "https://" + url.removeprefix("wss://")
3795
+ if url.startswith("ws://"):
3796
+ return "http://" + url.removeprefix("ws://")
3797
+ return url
3798
+
3799
+
3800
+ def _remove_room_listener(room: rtc.Room, event: str, listener) -> None:
3801
+ try:
3802
+ room.off(event, listener)
3803
+ except (AttributeError, ValueError):
3804
+ logger.debug("LiveKit listener was already removed", extra={"event": event})
3805
+
3806
+
3807
+ async def _close_agent_session(session: AgentSession, *, timeout: float) -> None:
3808
+ """Close a session without abandoning teardown on the event loop.
3809
+
3810
+ A shielded, timed-out ``aclose`` used to keep running after the case had
3811
+ returned. Repeating that in a soak test accumulated SDK activities until
3812
+ the guest process failed. Graceful close gets a bounded opportunity; after
3813
+ that, cancel and reap it because room and process teardown are independent.
3814
+ """
3815
+ close_session = getattr(session, "aclose", None)
3816
+ if close_session is None:
3817
+ session.shutdown(drain=False)
3818
+ return
3819
+ close_task = asyncio.create_task(close_session())
3820
+ try:
3821
+ await asyncio.wait_for(asyncio.shield(close_task), timeout=timeout)
3822
+ except asyncio.TimeoutError:
3823
+ close_task.cancel()
3824
+ try:
3825
+ await asyncio.wait_for(close_task, timeout=1.0)
3826
+ except (Exception, asyncio.CancelledError):
3827
+ if not close_task.done():
3828
+ close_task.add_done_callback(_consume_background_task_result)
3829
+ raise
3830
+
3831
+
3832
+ def _consume_background_task_result(task: asyncio.Task) -> None:
3833
+ try:
3834
+ task.result()
3835
+ except (Exception, asyncio.CancelledError):
3836
+ pass
3837
+
3838
+
3839
+ def _is_not_found(exc: Exception) -> bool:
3840
+ code = getattr(exc, "code", None)
3841
+ return str(getattr(code, "value", code)).lower() in {
3842
+ "not_found",
3843
+ "404",
3844
+ }
3845
+
3846
+
3847
+ def _record_cleanup_error(
3848
+ errors: list[str],
3849
+ exc: Exception,
3850
+ operation: str,
3851
+ run_id: str,
3852
+ test_case_id: str,
3853
+ ) -> None:
3854
+ errors.append(f"{operation}:{type(exc).__name__}")
3855
+ logger.error(
3856
+ "LiveKit cleanup operation failed",
3857
+ exc_info=redacted_exc_info(exc),
3858
+ extra={
3859
+ "run_id": run_id,
3860
+ "test_case_id": test_case_id,
3861
+ "operation": operation,
3862
+ "exception_type": type(exc).__name__,
3863
+ },
3864
+ )
3865
+
3866
+
3867
+ _LIVEKIT_INBOUND_TRUNK_ENV = "LIVEKIT_INBOUND_TRUNK_ID"
3868
+
3869
+
3870
+ def _safe_provider_error_details(
3871
+ exc: Exception, *, operation: str
3872
+ ) -> dict[str, object]:
3873
+ """Extract sanitized error attributes for report failures.
3874
+
3875
+ Never returns the exception message; only structural fields that are
3876
+ known to be safe from LiveKit/Twirp exception classes.
3877
+ """
3878
+
3879
+ code = getattr(exc, "code", None)
3880
+ if code is not None:
3881
+ code_value = getattr(code, "value", None)
3882
+ if code_value is None and not isinstance(code, (str, int)):
3883
+ code_value = str(code)
3884
+ else:
3885
+ code_value = code_value if code_value is not None else code
3886
+ else:
3887
+ code_value = None
3888
+ status = getattr(exc, "status", None) or getattr(exc, "status_code", None)
3889
+ details: dict[str, object] = {
3890
+ "operation": operation,
3891
+ "exception_type": type(exc).__name__,
3892
+ }
3893
+ if code_value is not None:
3894
+ details["provider_code"] = code_value
3895
+ if status is not None:
3896
+ try:
3897
+ details["http_status"] = int(status)
3898
+ except (TypeError, ValueError):
3899
+ details["http_status"] = str(status)
3900
+ metadata = getattr(exc, "metadata", None)
3901
+ if isinstance(metadata, dict):
3902
+ for key in ("sip_status_code", "sip_status", "sip-code"):
3903
+ value = metadata.get(key)
3904
+ if value is not None:
3905
+ details["sip_status_code"] = str(value)
3906
+ break
3907
+ return details
3908
+
3909
+
3910
+ async def _ensure_sip_inbound_dispatch(
3911
+ api_client: api.LiveKitAPI,
3912
+ *,
3913
+ transport: TelephonyTransport,
3914
+ room_name: str,
3915
+ ) -> tuple[str, bool]:
3916
+ """Return ``(sip_dispatch_rule_id, created_by_sdk)``.
3917
+
3918
+ When ``transport.dispatch_rule_name`` is supplied the SDK verifies
3919
+ the rule exists and reuses it. Otherwise the SDK provisions a
3920
+ per-run direct rule bound to ``LIVEKIT_INBOUND_TRUNK_ID`` that routes
3921
+ incoming calls into ``room_name`` — the same room the local
3922
+ simulator has already joined — and returns its id so the caller can
3923
+ tear it down.
3924
+ """
3925
+
3926
+ existing = await api_client.sip.list_sip_dispatch_rule(ListSIPDispatchRuleRequest())
3927
+ if transport.dispatch_rule_name:
3928
+ for rule in existing.items:
3929
+ if rule.name != transport.dispatch_rule_name:
3930
+ continue
3931
+ direct = (
3932
+ getattr(rule.rule, "dispatch_rule_direct", None) if rule.rule else None
3933
+ )
3934
+ direct_room = getattr(direct, "room_name", "") if direct is not None else ""
3935
+ if not direct_room:
3936
+ raise RuntimeError(
3937
+ "sip_inbound_rule_mismatch: "
3938
+ f"{transport.dispatch_rule_name} is not a direct rule"
3939
+ )
3940
+ if direct_room != room_name:
3941
+ raise RuntimeError(
3942
+ "sip_inbound_rule_mismatch: "
3943
+ f"{transport.dispatch_rule_name} targets a different room"
3944
+ )
3945
+ return rule.sip_dispatch_rule_id, False
3946
+ raise RuntimeError(f"sip_inbound_rule_missing: {transport.dispatch_rule_name}")
3947
+ trunk_id = os.environ.get(_LIVEKIT_INBOUND_TRUNK_ENV)
3948
+ if not trunk_id:
3949
+ raise RuntimeError(
3950
+ f"sip_inbound_trunk_missing: set {_LIVEKIT_INBOUND_TRUNK_ENV}"
3951
+ )
3952
+ for rule in existing.items:
3953
+ if trunk_id and trunk_id in rule.trunk_ids:
3954
+ raise RuntimeError(
3955
+ "sip_inbound_route_conflict: existing dispatch rule "
3956
+ f"{rule.sip_dispatch_rule_id} already covers this trunk"
3957
+ )
3958
+ rule_name = f"sim-inbound-{room_name[-24:]}"
3959
+ resp = await api_client.sip.create_sip_dispatch_rule(
3960
+ CreateSIPDispatchRuleRequest(
3961
+ rule=SIPDispatchRule(
3962
+ dispatch_rule_direct=SIPDispatchRuleDirect(
3963
+ room_name=room_name,
3964
+ ),
3965
+ ),
3966
+ trunk_ids=[trunk_id],
3967
+ hide_phone_number=False,
3968
+ name=rule_name,
3969
+ )
3970
+ )
3971
+ return resp.sip_dispatch_rule_id, True
3972
+
3973
+
3974
+ async def _ensure_room_absent(
3975
+ api_client: api.LiveKitAPI, room_name: str, *, poll_interval: float = 0.5
3976
+ ) -> int:
3977
+ """Make ``room_name`` absent before the caller (re-)creates it.
3978
+
3979
+ The pool's dispatch rule stays live for the whole run, so an inbound call
3980
+ can re-create this room between our own delete and our next poll — hence
3981
+ the re-delete inside the loop rather than a single delete-then-poll. No
3982
+ internal deadline: the caller bounds this with ``asyncio.wait_for``.
3983
+ """
3984
+
3985
+ try:
3986
+ await api_client.room.delete_room(api.DeleteRoomRequest(room=room_name))
3987
+ except Exception as exc: # noqa: BLE001
3988
+ if not _is_not_found(exc):
3989
+ raise
3990
+ polls = 0
3991
+ while True:
3992
+ resp = await api_client.room.list_rooms(api.ListRoomsRequest(names=[room_name]))
3993
+ polls += 1
3994
+ if not resp.rooms:
3995
+ return polls
3996
+ try:
3997
+ await api_client.room.delete_room(api.DeleteRoomRequest(room=room_name))
3998
+ except Exception as exc: # noqa: BLE001
3999
+ if not _is_not_found(exc):
4000
+ raise
4001
+ await asyncio.sleep(poll_interval if polls < 4 else 5.0)
4002
+
4003
+
4004
+ def _unexpected_participants(
4005
+ room, *, simulator_identity: str, recorder_identity: str
4006
+ ) -> set[str]:
4007
+ return {str(p.identity) for p in room.remote_participants.values()} - {
4008
+ simulator_identity,
4009
+ recorder_identity,
4010
+ }
4011
+
4012
+
4013
+ # LiveKit: the other party's number on a SIP participant (the caller, for an
4014
+ # inbound call). sip.trunkPhoneNumber is OUR number — never use it.
4015
+ _SIP_REMOTE_NUMBER_ATTRIBUTE = "sip.phoneNumber"
4016
+
4017
+
4018
+ def _number_digits(value) -> str:
4019
+ text = str(value or "")
4020
+ if text.startswith("sip:"):
4021
+ text = text[len("sip:") :]
4022
+ text = text.split("@", 1)[0]
4023
+ text = text.split(";", 1)[0]
4024
+ return re.sub(r"\D", "", text)
4025
+
4026
+
4027
+ def _caller_matches(attributes, expected: str | None) -> bool | None:
4028
+ if not expected:
4029
+ return None
4030
+ observed = attributes.get(_SIP_REMOTE_NUMBER_ATTRIBUTE) if attributes else None
4031
+ if not observed:
4032
+ return None
4033
+ expected_digits = _number_digits(expected)
4034
+ observed_digits = _number_digits(observed)
4035
+ if len(expected_digits) < 7 or len(observed_digits) < 7:
4036
+ return None
4037
+ longer, shorter = (
4038
+ (expected_digits, observed_digits)
4039
+ if len(expected_digits) >= len(observed_digits)
4040
+ else (observed_digits, expected_digits)
4041
+ )
4042
+ return longer.endswith(shorter)
4043
+
4044
+
4045
+ class _LeasedRoomCallerMismatch(Exception):
4046
+ """Module-private: raised by the leased-room caller check (step 4a), caught
4047
+ by the case's top-level try so the post-try recordings/evidence/metadata
4048
+ block still runs for a call that was billed and answered."""
4049
+
4050
+
4051
+ async def _delete_sip_dispatch_rule(api_client: api.LiveKitAPI, rule_id: str) -> None:
4052
+ await api_client.sip.delete_sip_dispatch_rule(
4053
+ DeleteSIPDispatchRuleRequest(sip_dispatch_rule_id=rule_id)
4054
+ )
4055
+
4056
+
4057
+ async def _collect_provider_evidence(
4058
+ *,
4059
+ config: ProviderEvidenceConfig,
4060
+ transport: TelephonyTransport,
4061
+ run_id: str,
4062
+ test_case_id: str,
4063
+ case_directory: Path,
4064
+ started_at: datetime,
4065
+ target: _TargetParticipant | None,
4066
+ provider_call_id_hint: str | None = None,
4067
+ provider_api_key: str | None = None,
4068
+ provider_api_base_url: str | None = None,
4069
+ termination_source: str | None = None,
4070
+ ) -> tuple[EvidenceSourceSummary | None, list[ArtifactManifestEntry]]:
4071
+ call_id_hint = provider_call_id_hint
4072
+ caller_phone = transport.sip_number if transport.kind == "sip_outbound" else None
4073
+ callee_phone = transport.sip_call_to if transport.kind == "sip_outbound" else None
4074
+ if target is not None:
4075
+ if call_id_hint is None and config.participant_attribute:
4076
+ call_id_hint = target.attributes.get(config.participant_attribute)
4077
+ caller_phone = caller_phone or (
4078
+ target.attributes.get("sip.from")
4079
+ or target.attributes.get("sip.fromUser")
4080
+ or target.attributes.get("sip.callerNumber")
4081
+ )
4082
+ callee_phone = callee_phone or (
4083
+ target.attributes.get("sip.to")
4084
+ or target.attributes.get("sip.toUser")
4085
+ or target.attributes.get("sip.calledNumber")
4086
+ )
4087
+ context = EvidenceContext(
4088
+ run_id=run_id,
4089
+ test_case_id=test_case_id,
4090
+ case_directory=case_directory,
4091
+ started_at=started_at,
4092
+ call_id_hint=call_id_hint,
4093
+ caller_phone=caller_phone,
4094
+ callee_phone=callee_phone,
4095
+ termination_source=termination_source,
4096
+ )
4097
+ try:
4098
+ if config.provider == "vapi":
4099
+ adapter = VapiEvidenceSource(
4100
+ config,
4101
+ api_key=provider_api_key,
4102
+ api_base_url=provider_api_base_url,
4103
+ )
4104
+ elif config.provider == "retell":
4105
+ adapter = RetellEvidenceSource(
4106
+ config,
4107
+ api_key=provider_api_key,
4108
+ api_base_url=provider_api_base_url,
4109
+ )
4110
+ else:
4111
+ raise ProviderConfigError(
4112
+ f"unsupported_provider_evidence: {config.provider}"
4113
+ )
4114
+ except ProviderConfigError as exc:
4115
+ summary = EvidenceSourceSummary(
4116
+ source_id=f"{config.provider}:unconfigured",
4117
+ adapter=config.provider,
4118
+ evidence_class=_EVIDENCE_PROVIDER_REPORTED,
4119
+ available=False,
4120
+ redactions=["auth", "phone_e164"],
4121
+ metadata={"provider": config.provider, "reason": str(exc)},
4122
+ )
4123
+ return summary, []
4124
+ try:
4125
+ await adapter.connect(context)
4126
+ result: ProviderFetchResult = await adapter.fetch_final()
4127
+ except Exception as exc: # noqa: BLE001 — provider failures are first-class evidence
4128
+ logger.warning(
4129
+ "Provider evidence adapter failed",
4130
+ exc_info=redacted_exc_info(exc),
4131
+ extra={
4132
+ "provider": config.provider,
4133
+ "run_id": run_id,
4134
+ "test_case_id": test_case_id,
4135
+ },
4136
+ )
4137
+ summary = EvidenceSourceSummary(
4138
+ source_id=f"{config.provider}:error",
4139
+ adapter=config.provider,
4140
+ evidence_class=_EVIDENCE_PROVIDER_REPORTED,
4141
+ available=False,
4142
+ redactions=["auth", "phone_e164"],
4143
+ metadata={
4144
+ "provider": config.provider,
4145
+ "reason": "adapter_exception",
4146
+ "exception_type": type(exc).__name__,
4147
+ },
4148
+ )
4149
+ return summary, []
4150
+ finally:
4151
+ try:
4152
+ await adapter.close()
4153
+ except Exception as exc: # noqa: BLE001
4154
+ logger.debug(
4155
+ "Provider evidence adapter close failed",
4156
+ extra={
4157
+ "provider": config.provider,
4158
+ "exception_type": type(exc).__name__,
4159
+ },
4160
+ )
4161
+ return result.summary, result.artifacts
4162
+
4163
+
4164
+ # Import lazily to avoid a module-import cycle with ProviderConfigError above.
4165
+ from fi.simulate.evidence.base import EvidenceClass as _EvidenceClass # noqa: E402
4166
+
4167
+ _EVIDENCE_PROVIDER_REPORTED = _EvidenceClass.PROVIDER_REPORTED