agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1440 @@
1
+ """The hosted lane's `CallRunner` — places one simulated LiveKit voice call and reports what
2
+ happened, satisfying `hosted_scheduler.CallRunner` exactly.
3
+
4
+ Three sub-systems (world-handle-interface.md, hosted-execution-seams.md v1.15 §2a):
5
+
6
+ 1. **Placing the call.** The customer agent is already running INSIDE the Daytona sandbox, as a
7
+ world process the bundle's provisioner spawned (`process_runtime.py`) and registered with
8
+ LiveKit cloud under `LIVEKIT_AGENT_NAME=agent-w{WORLD_INDEX}`-style identity. This runner never
9
+ starts or manages that process. It drives `SimulationRunner` IN-PROCESS with a
10
+ `SimulationSpec` built by `simulator_voice.simulation_spec`, the same builder the local lane
11
+ uses; only the value lookup differs, resolving from job config and the bundle's scenario
12
+ document rather than `HARNESS_*` env vars. Do not rebuild the spec here: the two lanes drifted
13
+ for exactly that reason. The local-only webhook/subprocess plumbing `run/call.py` and
14
+ `run/live.py` use is neither available nor appropriate in the guest.
15
+ 2. **Collecting evidence.** The bundle declares exactly one `runtime.evidence_seam`:
16
+ `http_tool` or `tool_trace`. `http_tool` has NO guest-side capture surface anywhere in this
17
+ repo today (see `_collect_http_tool_calls`'s docstring — a verified finding, not an assumption)
18
+ and is intentionally left returning zero calls rather than inventing a capture proxy.
19
+ `tool_trace` is read from the world's own postgres database against an unpinned, isolated
20
+ convention (see `_collect_tool_trace_calls`'s docstring). Either way, zero calls captured is
21
+ never fabricated into something else — the scheduler's own `evidence_missing` retry-once policy
22
+ is the contract-correct handling for "no evidence."
23
+ 3. **Uploading artifacts.** The transcript and any produced recordings are uploaded through the
24
+ adapter's `upload_artifact` (content-addressed, budget/level-gated, returns `None` on refusal —
25
+ never an exception) BEFORE this runner returns, so `CallOutcome.transcript_artifact`/
26
+ `recording_artifacts` only ever carry ids the platform has already acked.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import atexit
32
+ import asyncio
33
+ import gc
34
+ import json
35
+ import logging
36
+ import os
37
+ import stat
38
+ import tempfile
39
+ from dataclasses import dataclass, field, replace
40
+ from datetime import datetime, timezone
41
+ from pathlib import Path
42
+ from typing import Any, Awaitable, Callable, Mapping, Protocol
43
+
44
+ from fi import simulate
45
+ from fi.simulate.runtime import (
46
+ SimulationSpec,
47
+ new_run_id,
48
+ )
49
+ from fi.simulate.runtime.report import SimulationReport
50
+ from fi.simulate.runtime.run import TestCaseStatus
51
+ from fi.simulate.runtime.runner import SimulationRunner
52
+
53
+ from .background_noise import scenario_source
54
+ from .bundle_v2 import EvidenceSeam
55
+ from .hosted_scheduler import CallAborted, CallOutcome
56
+ from .hosted_scheduler import Scenario as HostedScenario
57
+ from .job import ExecutionMode, HarnessJob, ProviderExecutionMode
58
+ from .outbound import ArtifactKind, format_rfc3339_millis
59
+ from .process_runtime import EnvironmentRuntime
60
+ from .scenario import DEFAULT_VOICEMAIL_STYLE, voicemail_enabled
61
+ from .voicemail_audio import clip_for
62
+ from .simulator_voice import (
63
+ CLEANUP_TIMEOUT_SECONDS,
64
+ CONNECT_TIMEOUT_SECONDS,
65
+ READINESS_TIMEOUT_SECONDS,
66
+ caller_scenario,
67
+ simulation_spec,
68
+ simulator_definition,
69
+ )
70
+ from .world.errors import WorldUnavailable
71
+ from .world.runtime import Call
72
+
73
+ logger = logging.getLogger(__name__)
74
+
75
+ # --- credential aliases / config keys a voice job must carry ---
76
+
77
+ LIVEKIT_API_KEY_ALIAS = "LIVEKIT_API_KEY"
78
+ LIVEKIT_API_SECRET_ALIAS = "LIVEKIT_API_SECRET"
79
+ LIVEKIT_URL_ALIAS = "LIVEKIT_URL"
80
+ DEEPGRAM_API_KEY_ALIAS = "DEEPGRAM_API_KEY"
81
+ CARTESIA_API_KEY_ALIAS = "CARTESIA_API_KEY"
82
+ GEMINI_API_KEY_ALIAS = "GEMINI_API_KEY"
83
+ GOOGLE_API_KEY_ALIAS = "GOOGLE_API_KEY"
84
+ GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS = "GOOGLE_APPLICATION_CREDENTIALS_JSON"
85
+ GOOGLE_APPLICATION_CREDENTIALS_ALIAS = "GOOGLE_APPLICATION_CREDENTIALS"
86
+ GOOGLE_CLOUD_PROJECT_ALIAS = "GOOGLE_CLOUD_PROJECT"
87
+ GOOGLE_CLOUD_LOCATION_ALIAS = "GOOGLE_CLOUD_LOCATION"
88
+ GOOGLE_GENAI_USE_VERTEXAI_ALIAS = "GOOGLE_GENAI_USE_VERTEXAI"
89
+ OPENAI_API_KEY_ALIAS = "OPENAI_API_KEY"
90
+ VAPI_API_KEY_ALIAS = "VAPI_API_KEY"
91
+ RETELL_API_KEY_ALIAS = "RETELL_API_KEY"
92
+ SIMULATOR_LLM_PROVIDER_ALIAS = "SIMULATOR_LLM_PROVIDER"
93
+ SIMULATOR_LLM_MODEL_ALIAS = "SIMULATOR_LLM_MODEL"
94
+ SIMULATOR_STT_PROVIDER_ALIAS = "SIMULATOR_STT_PROVIDER"
95
+ SIMULATOR_STT_MODEL_ALIAS = "SIMULATOR_STT_MODEL"
96
+ SIMULATOR_TTS_PROVIDER_ALIAS = "SIMULATOR_TTS_PROVIDER"
97
+ SIMULATOR_TTS_MODEL_ALIAS = "SIMULATOR_TTS_MODEL"
98
+ BACKGROUND_NOISE_ALIAS = "ALK_BACKGROUND_NOISE"
99
+ BACKGROUND_NOISE_CATALOG_ALIAS = "ALK_BACKGROUND_NOISE_CATALOG"
100
+ BACKGROUND_NOISE_VOLUME_ALIAS = "HARNESS_BACKGROUND_NOISE_VOLUME"
101
+ CALL_DIRECTION_ALIAS = "ALK_CALL_DIRECTION"
102
+ VOICEMAIL_CLIP_ALIAS = "HARNESS_VOICEMAIL_CLIP"
103
+ VOICEMAIL_CLIP_TONE_ALIAS = "HARNESS_VOICEMAIL_CLIP_HAS_TONE"
104
+ VOICEMAIL_CLIP_TEXT_ALIAS = "HARNESS_VOICEMAIL_CLIP_TRANSCRIPT"
105
+ LIVEKIT_URL_CONFIG_KEY = "livekit_url"
106
+ CALL_TIMEOUT_CONFIG_KEY = "voice_call_timeout_seconds"
107
+
108
+ _SIMULATOR_PLATFORM_ALIAS_MAP = {
109
+ "SIMULATOR_LIVEKIT_URL": LIVEKIT_URL_ALIAS,
110
+ "SIMULATOR_LIVEKIT_API_KEY": LIVEKIT_API_KEY_ALIAS,
111
+ "SIMULATOR_LIVEKIT_API_SECRET": LIVEKIT_API_SECRET_ALIAS,
112
+ "SIMULATOR_DEEPGRAM_API_KEY": DEEPGRAM_API_KEY_ALIAS,
113
+ "SIMULATOR_CARTESIA_API_KEY": CARTESIA_API_KEY_ALIAS,
114
+ "SIMULATOR_GEMINI_API_KEY": GEMINI_API_KEY_ALIAS,
115
+ "SIMULATOR_GOOGLE_API_KEY": GOOGLE_API_KEY_ALIAS,
116
+ "SIMULATOR_GOOGLE_APPLICATION_CREDENTIALS_JSON": (
117
+ GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS
118
+ ),
119
+ "SIMULATOR_GOOGLE_CLOUD_PROJECT": GOOGLE_CLOUD_PROJECT_ALIAS,
120
+ "SIMULATOR_GOOGLE_CLOUD_LOCATION": GOOGLE_CLOUD_LOCATION_ALIAS,
121
+ "SIMULATOR_GOOGLE_GENAI_USE_VERTEXAI": GOOGLE_GENAI_USE_VERTEXAI_ALIAS,
122
+ "SIMULATOR_OPENAI_API_KEY": OPENAI_API_KEY_ALIAS,
123
+ }
124
+
125
+ _DEFAULT_CALL_TIMEOUT_SECONDS = 300.0
126
+
127
+ # sdk_voice.py::build_spec's own phase-overhead constants, reused verbatim so this runner's
128
+ # outer budget composes with the SDK's internal one the same way the local template does.
129
+ _RUN_SECONDS_PAD_SECONDS = 60.0
130
+ # Headroom beyond `spec.execution.timeout.run_seconds` -- SimulationRunner.run() already wraps
131
+ # `plugin.run(...)` in its OWN `asyncio.wait_for(..., timeout=spec.execution.timeout.run_seconds)`
132
+ # (runner.py) and catches that TimeoutError into a graceful `SimulationReport(status=TIMED_OUT)`.
133
+ # This runner's own outer wait_for must stay LARGER than that so the SDK's internal timeout fires
134
+ # first in the ordinary case; it only ever fires itself for a genuinely hung SDK (a real post-dial
135
+ # machinery failure) -- a runner-owned asyncio.wait_for as the last-resort bound.
136
+ _OUTER_WAIT_FOR_PAD_SECONDS = 60.0
137
+
138
+ # Unpinned by any contract and no producer exists yet. Isolated as one
139
+ # constant + two functions (`_clear_tool_trace_calls`, `_collect_tool_trace_calls`) so a real
140
+ # producer's disagreement on the name/shape is a one-line change.
141
+ _TOOL_TRACE_TABLE = "_alk_tool_trace"
142
+
143
+ _RESULT_TRUNCATE_CHARS = 2000
144
+
145
+ # Turns a timed-out call needs before it is worth grading rather than aborting. Low on purpose: the
146
+ # question is only whether a conversation happened at all.
147
+ _GRADEABLE_AFTER_TIMEOUT_TURNS = 4
148
+
149
+ # The real engine's zero-turn "agent joined but never spoke" failure codes (engines/livekit.py::
150
+ # _conversation_outcome) -- see `_translate_report`'s `is_silent_agent` gate for why these two, and
151
+ # only at zero turns, get mapped to a normal CallOutcome instead of a CallAborted.
152
+ _SILENT_AGENT_FAILURE_CODES = frozenset(
153
+ {"no_conversation", "conversation_silence_timeout"}
154
+ )
155
+ _CONVERSATION_STALL_FAILURE_CODES = frozenset(
156
+ {"conversation_silence_timeout", "conversation_stalled"}
157
+ )
158
+
159
+
160
+ def _attributed_stall(case: Any) -> tuple[str, str] | None:
161
+ """Attribute a speech stall from committed transcript turns, without guessing."""
162
+ if (
163
+ case.failure is None
164
+ or case.failure.code not in _CONVERSATION_STALL_FAILURE_CODES
165
+ ):
166
+ return None
167
+ if case.result is None or not case.result.messages:
168
+ return None
169
+ last = case.result.messages[-1]
170
+ if not isinstance(last, dict):
171
+ return None
172
+ role = str(last.get("role") or "").strip().lower()
173
+ content = str(last.get("content") or "").strip()
174
+ if role in {"user", "caller", "customer"}:
175
+ return (
176
+ "target_agent_stalled",
177
+ "Target agent produced no response after the caller's final transcribed turn",
178
+ )
179
+ if role in {"assistant", "agent"}:
180
+ if content and content[-1] not in ".?!":
181
+ return (
182
+ "target_agent_stalled",
183
+ "Target agent stopped mid-utterance and produced no further speech",
184
+ )
185
+ return (
186
+ "simulator_stalled",
187
+ "Simulated caller produced no response after the target agent's final turn",
188
+ )
189
+ return None
190
+
191
+
192
+ # --- collaborator seams (named, injectable test boundaries) -----------------------------------
193
+
194
+
195
+ class ArtifactUploader(Protocol):
196
+ """Narrow slice of `hosted_entrypoint.OutboundAdapter` -- avoids importing that module here
197
+ (it imports THIS module's factory to wire the real CallRunner; importing it back would be
198
+ circular)."""
199
+
200
+ async def upload_artifact(
201
+ self,
202
+ data: bytes,
203
+ *,
204
+ kind: ArtifactKind,
205
+ scenario_key: str | None = None,
206
+ deadline: float | None = None,
207
+ ) -> str | None: ...
208
+
209
+
210
+ PlaceCall = Callable[[SimulationSpec], Awaitable[SimulationReport]]
211
+
212
+
213
+ async def _default_place_call(spec: SimulationSpec) -> SimulationReport:
214
+ return await SimulationRunner().run(spec)
215
+
216
+
217
+ @dataclass(frozen=True)
218
+ class CallRunnerContext:
219
+ """Everything `hosted_entrypoint.py`'s `run_job` already has in scope by the wiring point
220
+ (~1662) that the real `CallRunnerImpl` needs but the bare `CallRunner` protocol signature
221
+ (`run(scenario, runtime)`) has no room to carry. Threaded through the EXTENDED
222
+ `build_call_runner(adapter, context)` seam."""
223
+
224
+ job: HarnessJob
225
+ bundle_dir: Path
226
+ work_directory: Path
227
+ evidence_seam: EvidenceSeam | None
228
+ target_provider_secret_values: Mapping[str, str]
229
+ attempt_number: int
230
+ source_directory: Path | None = None
231
+ simulator_provider_secret_values: Mapping[str, str] = field(default_factory=dict)
232
+
233
+
234
+ # --- pre-dial validation -----------------------------------------------------------------------
235
+
236
+
237
+ @dataclass(frozen=True)
238
+ class _MissingVoiceConfig:
239
+ aliases: tuple[str, ...]
240
+ config_keys: tuple[str, ...]
241
+
242
+ def message(self) -> str:
243
+ parts = []
244
+ if self.aliases:
245
+ parts.append("secrets=" + ",".join(self.aliases))
246
+ if self.config_keys:
247
+ parts.append("config=" + ",".join(self.config_keys))
248
+ return "voice_capability_unavailable: missing " + "; ".join(parts)
249
+
250
+
251
+ def _resolve_connector(
252
+ job: HarnessJob, target_provider_secret_values: Mapping[str, str]
253
+ ) -> str:
254
+ """Pin the job's transport connector from the credentials actually present.
255
+
256
+ A fresh one-shot ships ``job.json`` with ``connector="auto"``: the platform only writes the
257
+ authored connector back onto the job *after* authoring, by which time this guest has already
258
+ booted from the un-resolved payload. Mirror the platform rule so a LiveKit-credentialed
259
+ ``auto`` job dispatches to the target agent instead of the simulator lane with no identity.
260
+ """
261
+ connector = job.agent.connector.strip().lower()
262
+ if connector != "auto":
263
+ return connector
264
+ if job.agent.config.get(
265
+ LIVEKIT_URL_CONFIG_KEY
266
+ ) or target_provider_secret_values.get(LIVEKIT_URL_ALIAS):
267
+ return "livekit"
268
+ if target_provider_secret_values.get(VAPI_API_KEY_ALIAS):
269
+ return "vapi"
270
+ if target_provider_secret_values.get(RETELL_API_KEY_ALIAS):
271
+ return "retell"
272
+ return connector
273
+
274
+
275
+ def _check_config(
276
+ job: HarnessJob,
277
+ target_provider_secret_values: Mapping[str, str],
278
+ simulator_values: Mapping[str, str] | None = None,
279
+ ) -> _MissingVoiceConfig | None:
280
+ simulator_values = simulator_values or {}
281
+
282
+ def simulator_value(alias: str) -> str | None:
283
+ # Hosted runs supply platform-owned simulator credentials in the control process. The
284
+ # target-provider value remains a backwards-compatible fallback for local SDK callers.
285
+ return simulator_values.get(alias) or (
286
+ target_provider_secret_values.get(alias)
287
+ if job.execution is ExecutionMode.LOCAL
288
+ else None
289
+ )
290
+
291
+ config = job.agent.config
292
+ llm_provider = str(
293
+ config.get("simulator_llm_provider")
294
+ or simulator_value(SIMULATOR_LLM_PROVIDER_ALIAS)
295
+ or "google"
296
+ ).lower()
297
+ stt_provider = str(
298
+ config.get("simulator_stt_provider")
299
+ or simulator_value(SIMULATOR_STT_PROVIDER_ALIAS)
300
+ or "deepgram"
301
+ ).lower()
302
+ tts_provider = str(
303
+ config.get("simulator_tts_provider")
304
+ or simulator_value(SIMULATOR_TTS_PROVIDER_ALIAS)
305
+ or "deepgram"
306
+ ).lower()
307
+
308
+ connector = _resolve_connector(job, target_provider_secret_values)
309
+ livekit_values = (
310
+ target_provider_secret_values if connector == "livekit" else simulator_values
311
+ )
312
+ required = [LIVEKIT_API_KEY_ALIAS, LIVEKIT_API_SECRET_ALIAS]
313
+ if connector == "vapi":
314
+ required.append(VAPI_API_KEY_ALIAS)
315
+ elif connector == "retell":
316
+ required.append(RETELL_API_KEY_ALIAS)
317
+ if "deepgram" in {stt_provider, tts_provider}:
318
+ if not simulator_value(DEEPGRAM_API_KEY_ALIAS):
319
+ required.append(DEEPGRAM_API_KEY_ALIAS)
320
+
321
+ def credential(alias: str) -> str | None:
322
+ if alias in {LIVEKIT_API_KEY_ALIAS, LIVEKIT_API_SECRET_ALIAS}:
323
+ return livekit_values.get(alias)
324
+ if alias in {VAPI_API_KEY_ALIAS, RETELL_API_KEY_ALIAS}:
325
+ return target_provider_secret_values.get(alias)
326
+ return simulator_value(alias)
327
+
328
+ missing_aliases = [alias for alias in required if not credential(alias)]
329
+
330
+ if llm_provider == "google":
331
+ has_api_key = bool(
332
+ simulator_value(GEMINI_API_KEY_ALIAS)
333
+ or simulator_value(GOOGLE_API_KEY_ALIAS)
334
+ )
335
+ has_vertex_adc = bool(
336
+ (
337
+ simulator_value(GOOGLE_APPLICATION_CREDENTIALS_ALIAS)
338
+ or simulator_value(GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS)
339
+ )
340
+ and simulator_value(GOOGLE_CLOUD_PROJECT_ALIAS)
341
+ )
342
+ if not has_api_key and not has_vertex_adc:
343
+ missing_aliases.append(
344
+ f"{GEMINI_API_KEY_ALIAS}_or_{GOOGLE_API_KEY_ALIAS}_or_VERTEX_ADC"
345
+ )
346
+ elif llm_provider == "openai" and not simulator_value(OPENAI_API_KEY_ALIAS):
347
+ missing_aliases.append(OPENAI_API_KEY_ALIAS)
348
+
349
+ has_livekit_url = bool(
350
+ config.get(LIVEKIT_URL_CONFIG_KEY) or livekit_values.get(LIVEKIT_URL_ALIAS)
351
+ )
352
+ missing_config_keys = [] if has_livekit_url else [LIVEKIT_URL_CONFIG_KEY]
353
+ if not missing_aliases and not missing_config_keys:
354
+ return None
355
+ return _MissingVoiceConfig(tuple(missing_aliases), tuple(missing_config_keys))
356
+
357
+
358
+ def _canonical_simulator_secrets(values: Mapping[str, str]) -> dict[str, str]:
359
+ """Translate platform-only aliases into the names expected by simulator plugins."""
360
+ return {
361
+ _SIMULATOR_PLATFORM_ALIAS_MAP.get(alias, alias): value
362
+ for alias, value in values.items()
363
+ }
364
+
365
+
366
+ def _dispatch_agent_name(runtime: EnvironmentRuntime) -> str | None:
367
+ """The ONLY place this repo reads the dispatch-identity metadata key, so a
368
+ change to the key name/convention is a one-line adapt. The provisioner
369
+ mirrors the agent process's rendered LIVEKIT_AGENT_NAME here; a bundle
370
+ that declares none (or an ambiguous set) leaves the key absent and the
371
+ caller's typed `CallAborted` below fires."""
372
+ value = runtime.metadata.get("livekit_agent_name")
373
+ return value.strip() if isinstance(value, str) and value.strip() else None
374
+
375
+
376
+ # --- scenario document re-read (the _CompiledScenario the scheduler hands over carries no
377
+ # persona/instruction -- scenario_source.py:170-184's deliberately narrow Scenario-protocol
378
+ # shape) ------------------------------------------------------------------------------------
379
+
380
+
381
+ class _ScenarioDocumentUnavailable(RuntimeError):
382
+ pass
383
+
384
+
385
+ def _read_scenario_document(bundle_dir: Path, scenario_key: str) -> dict[str, Any]:
386
+ """Re-reads `scenarios/<folder>/scenario.json` from the bundle, matched by the document's OWN
387
+ `scenario_key` field -- never the folder name (`scenario_source.py`'s own convention; the two
388
+ are not guaranteed to match)."""
389
+ root = bundle_dir / "scenarios"
390
+ if not root.is_dir():
391
+ raise _ScenarioDocumentUnavailable(f"no {root} directory in this bundle")
392
+ try:
393
+ children = sorted(root.iterdir())
394
+ except OSError as exc:
395
+ raise _ScenarioDocumentUnavailable(f"cannot list {root}: {exc}") from exc
396
+ for child in children:
397
+ if not child.is_dir():
398
+ continue
399
+ doc_path = child / "scenario.json"
400
+ if not doc_path.is_file():
401
+ continue
402
+ try:
403
+ body = json.loads(doc_path.read_text(encoding="utf-8"))
404
+ except (OSError, ValueError):
405
+ continue
406
+ if isinstance(body, dict) and body.get("scenario_key") == scenario_key:
407
+ instruction = body.get("instruction")
408
+ if not isinstance(instruction, str) or not instruction.strip():
409
+ raise _ScenarioDocumentUnavailable(
410
+ f"{child.name}/scenario.json has no non-empty instruction"
411
+ )
412
+ return body
413
+ raise _ScenarioDocumentUnavailable(
414
+ f"no scenario.json under {root} carries scenario_key={scenario_key!r}"
415
+ )
416
+
417
+
418
+ # --- deterministic room naming (asserted verbatim by tests/harness/test_call_runner.py). WHY this
419
+ # is a PREFIX guarantee, not a full-match one: in managed room_mode, engines/livekit.py::
420
+ # _resolve_room_name appends its own `-{invocation_id}-{test_case_id[-12:]}` suffix unless
421
+ # `room_name_verbatim` is set (which this runner does not set) -- the scheme below still gives
422
+ # every call a unique, deterministic, greppable prefix; only the exact wire-level name is not this
423
+ # string verbatim. -------------------------------------------------------------------------------
424
+
425
+
426
+ def _room_name(
427
+ *, job_id: str, attempt_number: int, scenario_key: str, scenario_attempt: int
428
+ ) -> str:
429
+ return f"harness-{job_id[:8]}-a{attempt_number}-{scenario_key}-s{scenario_attempt}"
430
+
431
+
432
+ def _duration_ms(started_at: datetime, ended_at: datetime) -> int:
433
+ return max(0, int((ended_at - started_at).total_seconds() * 1000))
434
+
435
+
436
+ # --- SimulationSpec construction. Shared with the local lane through simulator_voice; only the
437
+ # value lookup is lane-specific. ---------------------------------------------------------------
438
+
439
+
440
+ def _dials_the_person(doc: dict[str, Any]) -> bool:
441
+ """Whether the agent places the call: scenario first, then the environment, then inbound."""
442
+ direction = str(
443
+ doc.get("call_direction") or os.environ.get(CALL_DIRECTION_ALIAS) or "inbound"
444
+ )
445
+ return direction.strip().lower() == "outbound"
446
+
447
+
448
+ def _build_spec(
449
+ *,
450
+ run_id: str,
451
+ room_name: str,
452
+ agent_name: str | None,
453
+ doc: Mapping[str, Any],
454
+ livekit_url: str,
455
+ call_timeout_seconds: float,
456
+ run_seconds: float,
457
+ recordings_root: Path,
458
+ simulator_config: Mapping[str, Any],
459
+ environ: Mapping[str, str],
460
+ connector: str = "livekit",
461
+ provider_target_id: str | None = None,
462
+ ) -> SimulationSpec:
463
+ """The hosted lane: values come from job config and the bundle's scenario document.
464
+
465
+ Lowercase job config wins; provider environment aliases are accepted for compatibility.
466
+ """
467
+
468
+ def setting(name: str) -> str:
469
+ return str(simulator_config.get(name.lower()) or environ.get(name) or "")
470
+
471
+ simulator = simulator_definition(setting, doc.get("persona"))
472
+ connector = connector.strip().lower()
473
+ provider_agent: simulate.AgentDefinition | None = None
474
+ if connector == "vapi":
475
+ if not provider_target_id:
476
+ raise ValueError("vapi_target_id_unavailable")
477
+ provider_agent = simulate.AgentDefinition(
478
+ name="harness-vapi-target",
479
+ system_prompt=str(
480
+ simulator_config.get("target_system_prompt")
481
+ or "Provider-hosted Vapi target under test."
482
+ ),
483
+ target={
484
+ "provider": "vapi",
485
+ "assistant_id": provider_target_id,
486
+ "api_base_url": str(
487
+ simulator_config.get("vapi_api_base_url") or "https://api.vapi.ai"
488
+ ),
489
+ "api_key_env": VAPI_API_KEY_ALIAS,
490
+ },
491
+ transport={"kind": "vapi_websocket"},
492
+ provider_evidence={
493
+ "provider": "vapi",
494
+ "call_id_source": "originator_response",
495
+ },
496
+ )
497
+ elif connector == "retell":
498
+ if not provider_target_id:
499
+ raise ValueError("retell_target_id_unavailable")
500
+ provider_agent = simulate.AgentDefinition(
501
+ name="harness-retell-target",
502
+ system_prompt=str(
503
+ simulator_config.get("target_system_prompt")
504
+ or "Provider-hosted Retell target under test."
505
+ ),
506
+ target={
507
+ "provider": "retell",
508
+ "agent_id": provider_target_id,
509
+ "api_url": str(
510
+ simulator_config.get("retell_api_url")
511
+ or "https://api.retellai.com/v2/create-web-call"
512
+ ),
513
+ "livekit_url": str(
514
+ simulator_config.get("retell_livekit_url")
515
+ or "wss://retell-ai-4ihahnq7.livekit.cloud"
516
+ ),
517
+ "api_key_env": RETELL_API_KEY_ALIAS,
518
+ },
519
+ transport={"kind": "retell_webcall"},
520
+ provider_evidence={
521
+ "provider": "retell",
522
+ "call_id_source": "originator_response",
523
+ },
524
+ )
525
+ # A provider-hosted agent owns termination and can legitimately finish an agent-first call
526
+ # after the fifth message: agent greeting, caller request, agent clarification, caller answer,
527
+ # agent confirmation followed by the provider's end-call tool. Requiring the simulator's
528
+ # sixth acknowledgement after Retell/Vapi has already disconnected misclassifies a complete
529
+ # call as infrastructure failure and prevents the tool trace from being graded. Native
530
+ # LiveKit keeps the stricter six-message floor because our simulator owns that hang-up path.
531
+ min_turn_messages = 5 if connector in {"vapi", "retell"} else 6
532
+ return simulation_spec(
533
+ run_id=run_id,
534
+ room_name=room_name,
535
+ agent_name=agent_name,
536
+ system_prompt=doc["instruction"],
537
+ livekit_url=livekit_url,
538
+ recording_dir=recordings_root / run_id / "recordings",
539
+ scenario=caller_scenario(
540
+ name=str(doc.get("scenario_key") or doc.get("name") or "harness-voice"),
541
+ persona=doc.get("persona"),
542
+ situation=doc["instruction"],
543
+ fixture=doc.get("fixture"),
544
+ tts_provider=simulator.tts.provider,
545
+ ),
546
+ simulator=simulator,
547
+ # An outbound agent dials; the person answers, so the caller opens.
548
+ direction="simulator_first" if _dials_the_person(doc) else "agent_first",
549
+ max_seconds=call_timeout_seconds,
550
+ min_turn_messages=min_turn_messages,
551
+ # Hosted targets can legitimately spend tens of seconds in a provider call or a tool
552
+ # round-trip after the conversation has begun. The previous 45-second value terminated
553
+ # an otherwise healthy LiveKit call at exactly the watchdog boundary. Keep a finite
554
+ # liveness guard, but align it with the engine's 60-second conversation-silence backstop.
555
+ agent_first_silence_seconds=60.0,
556
+ run_seconds=run_seconds,
557
+ agent_definition=provider_agent,
558
+ )
559
+
560
+
561
+ # --- evidence collection -----------------------------------------------------------------------
562
+
563
+
564
+ def _find_postgres_endpoint(runtime: EnvironmentRuntime) -> Any | None:
565
+ """Protocol-based lookup, matching `hosted_entrypoint.py::_find_postgres_endpoint`'s own
566
+ already-correct convention -- capability slugs are bundle-author-chosen (`build_endpoints`,
567
+ process_runtime.py:318-339), never a fixed key, so a hardcoded `endpoints["database"]` would
568
+ break for any bundle that names its capability slug differently. Re-implemented locally rather
569
+ than imported: importing from `hosted_entrypoint.py` here would be circular (it imports this
570
+ module's factory)."""
571
+ for endpoint in runtime.endpoints.values():
572
+ if endpoint.protocol == "postgres":
573
+ return endpoint
574
+ return None
575
+
576
+
577
+ def _collect_http_tool_calls(runtime: EnvironmentRuntime) -> tuple[Call, ...]:
578
+ """No guest-side capture surface exists anywhere in this repo for the
579
+ `http_tool` evidence seam. Verified, not assumed: `world/handle.py::HostedWorld.call()`
580
+ raises `WorldUnavailable` unconditionally with a docstring stating the wire format "is not
581
+ pinned anywhere in the contracts yet"; `process_runtime.py`'s own `provision()` signature
582
+ comment says "evidence-seam wiring is out of this phase's scope"; no `TOOLS_API_URL` wiring
583
+ exists in the hosted lane at all (the local lane's `ProvisionedWorld`/`TOOLS_API_URL` mechanism
584
+ lives in `provision.py`/`world/provisioned.py`, out of scope here and inapplicable to the guest
585
+ regardless). Deliberately stopped rather than inventing a capture proxy: a job whose bundle
586
+ declares `evidence_seam: http_tool` reads zero calls every time, which the scheduler's own
587
+ `evidence_missing` retry-once policy turns into the correct, honest outcome -- never a crash,
588
+ never fabricated evidence."""
589
+ del runtime
590
+ return ()
591
+
592
+
593
+ def _clear_tool_trace_calls(dsn: str) -> None:
594
+ """world-handle-interface.md: "setup's tool calls are NOT evidence (the runner clears them
595
+ before the call starts, as the local runner does)" -- the local runner's analog is
596
+ `world.calls = []` right before dialing (`run/simulation.py`). Best-effort: a missing table (no
597
+ producer yet) or any connection error is swallowed, never raised. Clearing
598
+ is housekeeping, not a correctness requirement, while nothing writes this table yet; once a
599
+ real producer lands this stops being a no-op automatically."""
600
+ try:
601
+ import psycopg
602
+
603
+ with psycopg.connect(dsn, autocommit=True, connect_timeout=5) as connection:
604
+ connection.execute(f'DELETE FROM "{_TOOL_TRACE_TABLE}"') # noqa: S608 - fixed identifier, no interpolated user input
605
+ except Exception as exc: # noqa: BLE001 - best-effort housekeeping only, never a call-blocking failure
606
+ # WHY: never log exc_info / str(exc) here -- a psycopg connection failure embeds the raw
607
+ # DSN (including the world DB password) in its own exception message; only the exception
608
+ # TYPE is safe for a local log line.
609
+ logger.debug(
610
+ "tool_trace clear skipped (table likely absent): %s", type(exc).__name__
611
+ )
612
+
613
+
614
+ def _collect_tool_trace_calls(runtime: EnvironmentRuntime) -> tuple[Call, ...]:
615
+ """`_alk_tool_trace`'s name and column shape are an isolated local
616
+ convention -- unpinned by any contract (the only harness-reserved table anywhere in this
617
+ repo is `_alk_conformance`, unrelated), no producer exists yet. Isolated in this one function
618
+ (+ `_clear_tool_trace_calls`) so a real producer's disagreement on the name/shape is a one-line
619
+ change. Any failure (missing table, connection refused, malformed row) degrades to `()` --
620
+ never a crash, never fabricated evidence, matching `_collect_http_tool_calls`'s stopped
621
+ behavior above."""
622
+ endpoint = _find_postgres_endpoint(runtime)
623
+ if endpoint is None:
624
+ return ()
625
+ try:
626
+ import psycopg
627
+
628
+ with psycopg.connect(
629
+ endpoint.address,
630
+ autocommit=True,
631
+ connect_timeout=5,
632
+ options="-c default_transaction_read_only=on",
633
+ ) as connection:
634
+ cursor = connection.execute(
635
+ f'SELECT name, arguments, result, ok, error, at FROM "{_TOOL_TRACE_TABLE}" ' # noqa: S608
636
+ "ORDER BY at ASC"
637
+ )
638
+ rows = cursor.fetchall()
639
+ columns = [description[0] for description in cursor.description or []]
640
+ except Exception as exc: # noqa: BLE001 - missing table / connection failure -> no evidence, not a crash
641
+ # WHY: same DSN-in-exception-message risk as `_clear_tool_trace_calls` above -- log only
642
+ # the exception TYPE, never exc_info/str(exc), which can carry the world DB password.
643
+ logger.debug(
644
+ "tool_trace read failed; treating as no evidence: %s", type(exc).__name__
645
+ )
646
+ return ()
647
+
648
+ calls: list[Call] = []
649
+ for row in rows:
650
+ record = dict(zip(columns, row, strict=True))
651
+ name = record.get("name")
652
+ if not isinstance(name, str) or not name:
653
+ continue
654
+ arguments = record.get("arguments")
655
+ if not isinstance(arguments, dict):
656
+ arguments = {}
657
+ ok = bool(record.get("ok", True))
658
+ raw_result = record.get("result")
659
+ if isinstance(raw_result, str):
660
+ result: Any = _truncate(raw_result)
661
+ else:
662
+ # Already parsed JSON (dict/list/etc, psycopg's own jsonb decoding) -- per
663
+ # world-handle-interface.md, only the STRING form is truncated at 2000 chars.
664
+ result = raw_result
665
+ error = _truncate(str(record.get("error") or ""))
666
+ raw_at = record.get("at")
667
+ at = float(raw_at) if isinstance(raw_at, (int, float)) else 0.0
668
+ calls.append(
669
+ Call(
670
+ name=name,
671
+ arguments=arguments,
672
+ result=result,
673
+ ok=ok,
674
+ error=error,
675
+ refused=not ok,
676
+ at=at,
677
+ )
678
+ )
679
+ return tuple(calls)
680
+
681
+
682
+ def _tool_trace_file(runtime: EnvironmentRuntime) -> Path | None:
683
+ raw = runtime.metadata.get("tool_trace_path")
684
+ return Path(raw) if isinstance(raw, str) and raw.strip() else None
685
+
686
+
687
+ def _clear_file_tool_calls(runtime: EnvironmentRuntime) -> None:
688
+ path = _tool_trace_file(runtime)
689
+ if path is None:
690
+ return
691
+ try:
692
+ path.unlink(missing_ok=True)
693
+ except OSError:
694
+ logger.debug(
695
+ "file tool_trace clear failed; continuing without blocking the call"
696
+ )
697
+
698
+
699
+ def _collect_file_tool_calls(runtime: EnvironmentRuntime) -> tuple[Call, ...]:
700
+ path = _tool_trace_file(runtime)
701
+ if path is None or not path.is_file():
702
+ return ()
703
+ calls: list[Call] = []
704
+ try:
705
+ lines = path.read_text(encoding="utf-8").splitlines()
706
+ except OSError:
707
+ return ()
708
+ for line in lines:
709
+ try:
710
+ record = json.loads(line)
711
+ except ValueError:
712
+ continue
713
+ if not isinstance(record, dict):
714
+ continue
715
+ name = record.get("name")
716
+ if not isinstance(name, str) or not name:
717
+ continue
718
+ arguments = record.get("arguments")
719
+ if isinstance(arguments, str):
720
+ try:
721
+ arguments = json.loads(arguments)
722
+ except ValueError:
723
+ arguments = {"raw": arguments}
724
+ if not isinstance(arguments, dict):
725
+ arguments = {}
726
+ is_error = bool(record.get("is_error", False))
727
+ output = record.get("output")
728
+ calls.append(
729
+ Call(
730
+ name=name,
731
+ arguments=arguments,
732
+ result=None if is_error else output,
733
+ ok=not is_error,
734
+ error=str(output) if is_error and output is not None else None,
735
+ )
736
+ )
737
+ return tuple(calls)
738
+
739
+
740
+ def _collect_provider_tool_calls(case: Any) -> tuple[Call, ...]:
741
+ """Translate provider-reported tool evidence into scheduler calls.
742
+
743
+ LiveKit's legacy report conversion stores per-case evidence in
744
+ ``result.metadata.evidence``; canonical reports may populate
745
+ ``case.evidence`` directly. Accept both shapes so hosted execution is not
746
+ coupled to the report representation.
747
+ """
748
+ sources: list[Any] = list(getattr(case, "evidence", None) or [])
749
+ result = getattr(case, "result", None)
750
+ result_metadata = getattr(result, "metadata", None)
751
+ if isinstance(result_metadata, Mapping):
752
+ embedded = result_metadata.get("evidence")
753
+ if isinstance(embedded, list):
754
+ sources.extend(embedded)
755
+
756
+ calls: list[Call] = []
757
+ for source in sources:
758
+ if hasattr(source, "model_dump"):
759
+ source = source.model_dump(mode="json", exclude_none=True)
760
+ if not isinstance(source, Mapping):
761
+ continue
762
+ metadata = source.get("metadata")
763
+ if not isinstance(metadata, Mapping):
764
+ continue
765
+ raw_calls = metadata.get("tool_calls")
766
+ if not isinstance(raw_calls, list):
767
+ continue
768
+ for raw in raw_calls:
769
+ if not isinstance(raw, Mapping):
770
+ continue
771
+ name = raw.get("name")
772
+ if not isinstance(name, str) or not name:
773
+ continue
774
+ arguments: Any = raw.get("arguments")
775
+ if isinstance(arguments, str):
776
+ try:
777
+ arguments = json.loads(arguments)
778
+ except ValueError:
779
+ arguments = {"raw": arguments}
780
+ if not isinstance(arguments, dict):
781
+ arguments = {}
782
+ ok = bool(raw.get("ok", True))
783
+ raw_at = raw.get("at")
784
+ at = float(raw_at) if isinstance(raw_at, (int, float)) else 0.0
785
+ calls.append(
786
+ Call(
787
+ name=name,
788
+ arguments=arguments,
789
+ result=raw.get("result") if ok else None,
790
+ ok=ok,
791
+ error=str(raw.get("error") or ""),
792
+ refused=not ok,
793
+ at=at,
794
+ )
795
+ )
796
+ return tuple(calls)
797
+
798
+
799
+ def _truncate(value: str, *, limit: int = _RESULT_TRUNCATE_CHARS) -> str:
800
+ return value if len(value) <= limit else value[:limit]
801
+
802
+
803
+ def _materialize_vertex_adc(
804
+ secret_values: Mapping[str, str],
805
+ work_directory: Path,
806
+ environ: dict[str, str],
807
+ ) -> Path | None:
808
+ """Materialize caller-lane Vertex credentials for Google ADC.
809
+
810
+ API-key auth needs no file. For Vertex, GOOGLE_APPLICATION_CREDENTIALS_JSON is resolved from
811
+ the platform vault and written mode-0600 under the job work directory; the sandbox is
812
+ ephemeral and the file is removed when the guest exits/deletes.
813
+ """
814
+ raw = secret_values.get(GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS)
815
+ if not raw or environ.get(GOOGLE_APPLICATION_CREDENTIALS_ALIAS):
816
+ return None
817
+ try:
818
+ parsed = json.loads(raw)
819
+ except json.JSONDecodeError as exc:
820
+ raise CallAborted(
821
+ "voice_capability_unavailable: GOOGLE_APPLICATION_CREDENTIALS_JSON is invalid"
822
+ ) from exc
823
+ if not isinstance(parsed, dict):
824
+ raise CallAborted(
825
+ "voice_capability_unavailable: GOOGLE_APPLICATION_CREDENTIALS_JSON must be an object"
826
+ )
827
+ credential_dir = work_directory / ".caller-credentials"
828
+ credential_dir.mkdir(parents=True, exist_ok=True)
829
+ fd, path_text = tempfile.mkstemp(
830
+ prefix="google-", suffix=".json", dir=credential_dir
831
+ )
832
+ path = Path(path_text)
833
+ try:
834
+ os.write(fd, raw.encode("utf-8"))
835
+ finally:
836
+ os.close(fd)
837
+ path.chmod(stat.S_IRUSR | stat.S_IWUSR)
838
+ environ[GOOGLE_APPLICATION_CREDENTIALS_ALIAS] = str(path)
839
+ return path
840
+
841
+
842
+ # --- the runner ----------------------------------------------------------------------------
843
+
844
+
845
+ class CallRunnerImpl:
846
+ """Satisfies `hosted_scheduler.CallRunner`. See the module docstring for the three
847
+ sub-systems this class implements."""
848
+
849
+ def __init__(
850
+ self,
851
+ adapter: ArtifactUploader,
852
+ context: CallRunnerContext,
853
+ *,
854
+ place_call: PlaceCall | None = None,
855
+ environ: dict[str, str] | None = None,
856
+ ) -> None:
857
+ self._adapter = adapter
858
+ self._context = context
859
+ self._place_call = place_call or _default_place_call
860
+ simulator_secret_values = _canonical_simulator_secrets(
861
+ context.simulator_provider_secret_values
862
+ )
863
+ # Local SDK runs remain BYOK and historically carry simulator keys in the one local
864
+ # target map. Hosted runs deliberately do not fall back: their simulator credentials must
865
+ # come from platform configuration and must not be confused with customer-agent keys.
866
+ if context.job.execution is ExecutionMode.LOCAL:
867
+ for alias in (
868
+ LIVEKIT_URL_ALIAS,
869
+ LIVEKIT_API_KEY_ALIAS,
870
+ LIVEKIT_API_SECRET_ALIAS,
871
+ DEEPGRAM_API_KEY_ALIAS,
872
+ CARTESIA_API_KEY_ALIAS,
873
+ GEMINI_API_KEY_ALIAS,
874
+ GOOGLE_API_KEY_ALIAS,
875
+ GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS,
876
+ GOOGLE_CLOUD_PROJECT_ALIAS,
877
+ GOOGLE_CLOUD_LOCATION_ALIAS,
878
+ GOOGLE_GENAI_USE_VERTEXAI_ALIAS,
879
+ OPENAI_API_KEY_ALIAS,
880
+ SIMULATOR_LLM_PROVIDER_ALIAS,
881
+ SIMULATOR_LLM_MODEL_ALIAS,
882
+ SIMULATOR_STT_PROVIDER_ALIAS,
883
+ SIMULATOR_STT_MODEL_ALIAS,
884
+ SIMULATOR_TTS_PROVIDER_ALIAS,
885
+ SIMULATOR_TTS_MODEL_ALIAS,
886
+ ):
887
+ if alias not in simulator_secret_values:
888
+ value = context.target_provider_secret_values.get(alias)
889
+ if value:
890
+ simulator_secret_values[alias] = value
891
+ # WHY: the underlying LiveKit engine reads these directly via `os.environ.get(...)` deep
892
+ # inside `engines/livekit.py` / `livekit_models.py` -- they are NOT `SimulationSpec`
893
+ # fields, so there is no other way to hand them over. Exported ONCE here, at construction,
894
+ # not per-call: the values are job-level (the same secret for every scenario/attempt on
895
+ # this job) and W>1 means each world's CallRunner.run() executes inside this SAME guest
896
+ # process but against a per-world sandboxed agent process reached over the network; no
897
+ # other in-process worker races this job-level environment.
898
+ target_environ = os.environ if environ is None else environ
899
+ connector = _resolve_connector(
900
+ context.job, context.target_provider_secret_values
901
+ )
902
+ target_aliases = [VAPI_API_KEY_ALIAS, RETELL_API_KEY_ALIAS]
903
+ if connector == "livekit":
904
+ target_aliases.extend(
905
+ [LIVEKIT_API_KEY_ALIAS, LIVEKIT_API_SECRET_ALIAS, LIVEKIT_URL_ALIAS]
906
+ )
907
+ for alias in target_aliases:
908
+ value = context.target_provider_secret_values.get(alias)
909
+ if value:
910
+ target_environ[alias] = value
911
+ for alias in (
912
+ LIVEKIT_URL_ALIAS,
913
+ LIVEKIT_API_KEY_ALIAS,
914
+ LIVEKIT_API_SECRET_ALIAS,
915
+ DEEPGRAM_API_KEY_ALIAS,
916
+ CARTESIA_API_KEY_ALIAS,
917
+ GEMINI_API_KEY_ALIAS,
918
+ GOOGLE_API_KEY_ALIAS,
919
+ GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS,
920
+ GOOGLE_CLOUD_PROJECT_ALIAS,
921
+ GOOGLE_CLOUD_LOCATION_ALIAS,
922
+ GOOGLE_GENAI_USE_VERTEXAI_ALIAS,
923
+ OPENAI_API_KEY_ALIAS,
924
+ SIMULATOR_LLM_PROVIDER_ALIAS,
925
+ SIMULATOR_LLM_MODEL_ALIAS,
926
+ SIMULATOR_STT_PROVIDER_ALIAS,
927
+ SIMULATOR_STT_MODEL_ALIAS,
928
+ SIMULATOR_TTS_PROVIDER_ALIAS,
929
+ SIMULATOR_TTS_MODEL_ALIAS,
930
+ BACKGROUND_NOISE_ALIAS,
931
+ BACKGROUND_NOISE_CATALOG_ALIAS,
932
+ BACKGROUND_NOISE_VOLUME_ALIAS,
933
+ ):
934
+ value = simulator_secret_values.get(alias)
935
+ if value:
936
+ # Platform-owned simulator credentials already present in the hosted control
937
+ # process win. Target credentials are retained only as the local-SDK fallback.
938
+ target_environ.setdefault(alias, value)
939
+ self._environ = target_environ
940
+ self._adc_path = _materialize_vertex_adc(
941
+ simulator_secret_values,
942
+ context.work_directory,
943
+ target_environ,
944
+ )
945
+ atexit.register(self._cleanup_credentials)
946
+ self._livekit_url = str(
947
+ context.job.agent.config.get(LIVEKIT_URL_CONFIG_KEY)
948
+ or (
949
+ context.target_provider_secret_values.get(LIVEKIT_URL_ALIAS)
950
+ if connector == "livekit"
951
+ else simulator_secret_values.get(LIVEKIT_URL_ALIAS)
952
+ )
953
+ or ""
954
+ )
955
+ self._missing_config = _check_config(
956
+ context.job,
957
+ context.target_provider_secret_values,
958
+ simulator_secret_values,
959
+ )
960
+ self._scenario_attempt_counts: dict[str, int] = {}
961
+ self._closed = False
962
+
963
+ def _cleanup_credentials(self) -> None:
964
+ if self._adc_path is None:
965
+ return
966
+ try:
967
+ self._adc_path.unlink(missing_ok=True)
968
+ except OSError:
969
+ pass
970
+ if self._environ.get(GOOGLE_APPLICATION_CREDENTIALS_ALIAS) == str(
971
+ self._adc_path
972
+ ):
973
+ self._environ.pop(GOOGLE_APPLICATION_CREDENTIALS_ALIAS, None)
974
+ self._adc_path = None
975
+
976
+ async def close(self) -> None:
977
+ """Release job-scoped resources before ``asyncio.run`` closes its event loop.
978
+
979
+ LiveKit's Python objects own native FFI handles and several of them participate in
980
+ reference cycles. Leaving those cycles to interpreter shutdown lets their finalizers run
981
+ after LiveKit's callback loop has closed; sufficiently long jobs then abort in the native
982
+ FFI teardown even though every call and artifact already completed. Collect on the event
983
+ loop thread and yield twice so queued FFI callbacks drain while their loop is still valid.
984
+ """
985
+ if self._closed:
986
+ return
987
+ self._closed = True
988
+ self._cleanup_credentials()
989
+ atexit.unregister(self._cleanup_credentials)
990
+ gc.collect()
991
+ # Native RTC shutdown is not synchronous with the Python objects that requested it.
992
+ # Give finalizers and already-enqueued disconnect/drop-handle callbacks real scheduling
993
+ # windows while the loop is still alive. Do not mutate LiveKit's private FFI subscriber
994
+ # list here: a subscriber is owned by its AudioStream task, and removing its queue behind
995
+ # that task's back produces stranded coroutines (observed after a 50-call soak as
996
+ # ``cannot reuse already awaited coroutine``). Deterministic collection on the live loop
997
+ # addresses the shutdown-order problem without violating stream ownership.
998
+ await asyncio.sleep(0.25)
999
+ gc.collect()
1000
+ await asyncio.sleep(0.25)
1001
+
1002
+ async def run(
1003
+ self,
1004
+ scenario: HostedScenario,
1005
+ runtime: EnvironmentRuntime,
1006
+ *,
1007
+ world: Any | None = None,
1008
+ ) -> CallOutcome:
1009
+ del world # Voice tools cross the declared evidence seam; they are not response-carried.
1010
+ if self._missing_config is not None:
1011
+ # Pre-dial: dialing never starts, so no partial -- and never `WorldUnavailable` (that
1012
+ # code is reserved by the contract for a world-level capability mismatch, not a
1013
+ # job-level voice config gap).
1014
+ raise CallAborted(self._missing_config.message())
1015
+
1016
+ connector = _resolve_connector(
1017
+ self._context.job, self._context.target_provider_secret_values
1018
+ )
1019
+ agent_name = _dispatch_agent_name(runtime) if connector == "livekit" else None
1020
+ if connector == "livekit" and agent_name is None:
1021
+ raise CallAborted(
1022
+ "voice_dispatch_identity_unavailable: runtime.metadata['livekit_agent_name'] is "
1023
+ f"not set for world {runtime.world_index}"
1024
+ )
1025
+
1026
+ try:
1027
+ doc = _read_scenario_document(
1028
+ self._context.bundle_dir, scenario.scenario_key
1029
+ )
1030
+ except _ScenarioDocumentUnavailable as exc:
1031
+ raise CallAborted(f"voice_scenario_document_unavailable: {exc}") from exc
1032
+
1033
+ scenario_attempt = (
1034
+ self._scenario_attempt_counts.get(scenario.scenario_key, 0) + 1
1035
+ )
1036
+ self._scenario_attempt_counts[scenario.scenario_key] = scenario_attempt
1037
+ room_name = _room_name(
1038
+ job_id=self._context.job.job_id,
1039
+ attempt_number=self._context.attempt_number,
1040
+ scenario_key=scenario.scenario_key,
1041
+ scenario_attempt=scenario_attempt,
1042
+ )
1043
+
1044
+ raw_timeout = self._context.job.agent.config.get(CALL_TIMEOUT_CONFIG_KEY)
1045
+ call_timeout_seconds = (
1046
+ float(raw_timeout)
1047
+ if isinstance(raw_timeout, (int, float))
1048
+ else _DEFAULT_CALL_TIMEOUT_SECONDS
1049
+ )
1050
+ run_seconds = (
1051
+ call_timeout_seconds
1052
+ + CONNECT_TIMEOUT_SECONDS
1053
+ + READINESS_TIMEOUT_SECONDS
1054
+ + CLEANUP_TIMEOUT_SECONDS
1055
+ + _RUN_SECONDS_PAD_SECONDS
1056
+ )
1057
+
1058
+ # The engine reads this from the environment at call time, so it is set per
1059
+ # scenario and cleared otherwise rather than leaking into the next call.
1060
+ noise = scenario_source(
1061
+ doc.get("background_noise"),
1062
+ doc.get("fixture"),
1063
+ seed=str(doc.get("name") or ""),
1064
+ )
1065
+ if noise:
1066
+ self._environ["HARNESS_BACKGROUND_NOISE"] = noise
1067
+ else:
1068
+ self._environ.pop("HARNESS_BACKGROUND_NOISE", None)
1069
+
1070
+ # Read the same way and for the same reason as the noise source above: the simulator's
1071
+ # instructions are built deep inside simulator_definition, which sees the environment and
1072
+ # not this scenario. Set per scenario and cleared otherwise so one outbound scenario cannot
1073
+ # frame the next inbound one.
1074
+ # A scenario that names its own direction wins. Otherwise the contract's, which the
1075
+ # understand stage read off the agent's own instructions and `hosted_entrypoint` puts here
1076
+ # for this process. Not an operator setting: whether an agent places calls or answers them
1077
+ # is a fact about the agent, so there is nothing for a run to choose.
1078
+ direction = (
1079
+ str(
1080
+ doc.get("call_direction")
1081
+ or os.environ.get(CALL_DIRECTION_ALIAS)
1082
+ or "inbound"
1083
+ )
1084
+ .strip()
1085
+ .lower()
1086
+ )
1087
+ if direction == "outbound":
1088
+ self._environ["HARNESS_CALL_DIRECTION"] = direction
1089
+ awareness = str(doc.get("caller_awareness") or "").strip().lower()
1090
+ if awareness:
1091
+ self._environ["HARNESS_CALLER_AWARENESS"] = awareness
1092
+ else:
1093
+ self._environ.pop("HARNESS_CALLER_AWARENESS", None)
1094
+ # Cleared otherwise, so one voicemail scenario cannot silence the next caller.
1095
+ if (
1096
+ voicemail_enabled()
1097
+ and str(doc.get("answered_by") or "").strip().lower() == "voicemail"
1098
+ ):
1099
+ self._environ["HARNESS_ANSWERED_BY"] = "voicemail"
1100
+ # Which kind of mailbox, which decides the greeting and whether a tone follows it.
1101
+ style = str(doc.get("voicemail_style") or "").strip().lower()
1102
+ if style:
1103
+ self._environ["HARNESS_VOICEMAIL_STYLE"] = style
1104
+ else:
1105
+ self._environ.pop("HARNESS_VOICEMAIL_STYLE", None)
1106
+ # A recorded greeting where the catalogue has one for this style AND language. It
1107
+ # replaces the spoken greeting rather than joining it.
1108
+ languages = doc.get("languages") or []
1109
+ chosen = clip_for(
1110
+ style or DEFAULT_VOICEMAIL_STYLE,
1111
+ str(languages[0]) if languages else "",
1112
+ )
1113
+ if chosen:
1114
+ self._environ[VOICEMAIL_CLIP_ALIAS] = chosen["source"]
1115
+ self._environ[VOICEMAIL_CLIP_TONE_ALIAS] = (
1116
+ "1" if chosen["has_tone"] else "0"
1117
+ )
1118
+ if chosen.get("transcript"):
1119
+ self._environ[VOICEMAIL_CLIP_TEXT_ALIAS] = chosen["transcript"]
1120
+ else:
1121
+ self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
1122
+ else:
1123
+ self._environ.pop(VOICEMAIL_CLIP_ALIAS, None)
1124
+ self._environ.pop(VOICEMAIL_CLIP_TONE_ALIAS, None)
1125
+ self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
1126
+ else:
1127
+ self._environ.pop("HARNESS_ANSWERED_BY", None)
1128
+ self._environ.pop("HARNESS_VOICEMAIL_STYLE", None)
1129
+ self._environ.pop(VOICEMAIL_CLIP_ALIAS, None)
1130
+ self._environ.pop(VOICEMAIL_CLIP_TONE_ALIAS, None)
1131
+ self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
1132
+ else:
1133
+ self._environ.pop("HARNESS_CALL_DIRECTION", None)
1134
+ self._environ.pop("HARNESS_CALLER_AWARENESS", None)
1135
+ self._environ.pop("HARNESS_ANSWERED_BY", None)
1136
+ self._environ.pop("HARNESS_VOICEMAIL_STYLE", None)
1137
+ self._environ.pop(VOICEMAIL_CLIP_ALIAS, None)
1138
+ self._environ.pop(VOICEMAIL_CLIP_TONE_ALIAS, None)
1139
+ self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
1140
+
1141
+ provider_target_key = {"vapi": "assistant_id", "retell": "agent_id"}.get(
1142
+ connector
1143
+ )
1144
+ provider_target_id: str | None = None
1145
+ if provider_target_key and self._context.job.agent.mode in {
1146
+ None,
1147
+ ProviderExecutionMode.CONNECT_ONLY,
1148
+ }:
1149
+ provider_target_id = str(
1150
+ self._context.job.agent.config.get(provider_target_key) or ""
1151
+ ).strip()
1152
+ if provider_target_key and self._context.job.agent.mode not in {
1153
+ None,
1154
+ ProviderExecutionMode.CONNECT_ONLY,
1155
+ }:
1156
+ dynamic_target = runtime.metadata.get("provider_target_id")
1157
+ provider_target_id = (
1158
+ dynamic_target.strip()
1159
+ if isinstance(dynamic_target, str) and dynamic_target.strip()
1160
+ else None
1161
+ )
1162
+
1163
+ spec = _build_spec(
1164
+ run_id=new_run_id(),
1165
+ room_name=room_name,
1166
+ connector=connector,
1167
+ agent_name=agent_name,
1168
+ provider_target_id=provider_target_id,
1169
+ doc=doc,
1170
+ simulator_config=self._context.job.agent.config,
1171
+ environ=self._environ,
1172
+ livekit_url=self._livekit_url,
1173
+ call_timeout_seconds=call_timeout_seconds,
1174
+ run_seconds=run_seconds,
1175
+ recordings_root=self._context.work_directory / "voice-calls",
1176
+ )
1177
+
1178
+ if self._context.evidence_seam is EvidenceSeam.TOOL_TRACE:
1179
+ _clear_file_tool_calls(runtime)
1180
+ endpoint = _find_postgres_endpoint(runtime)
1181
+ if endpoint is not None:
1182
+ _clear_tool_trace_calls(endpoint.address)
1183
+
1184
+ started_at = datetime.now(timezone.utc)
1185
+ outer_timeout = run_seconds + _OUTER_WAIT_FOR_PAD_SECONDS
1186
+ try:
1187
+ report = await asyncio.wait_for(
1188
+ self._place_call(spec), timeout=outer_timeout
1189
+ )
1190
+ except asyncio.CancelledError:
1191
+ raise
1192
+ except asyncio.TimeoutError as exc:
1193
+ raise CallAborted(
1194
+ "voice_call_runner_timeout: place_call exceeded its outer budget "
1195
+ f"({outer_timeout:.0f}s)",
1196
+ partial=self._timing_only_outcome(started_at),
1197
+ ) from exc
1198
+ except Exception as exc: # noqa: BLE001 - post-dial machinery failure, never let it escape raw
1199
+ raise CallAborted(
1200
+ f"voice_call_runner_crashed: {type(exc).__name__}: {exc}",
1201
+ partial=self._timing_only_outcome(started_at),
1202
+ ) from exc
1203
+
1204
+ try:
1205
+ return await self._translate_report(
1206
+ report,
1207
+ runtime=runtime,
1208
+ scenario_key=scenario.scenario_key,
1209
+ started_at=started_at,
1210
+ )
1211
+ except (CallAborted, WorldUnavailable):
1212
+ # `_translate_report`'s own typed control-flow (non-completed status, no test case,
1213
+ # agent-never-joined) -- never re-wrap an intentional abort.
1214
+ raise
1215
+ except Exception as exc: # noqa: BLE001 - a transcript/recording read or upload surprise
1216
+ # must never lose the timing this call already measured (the receipt's `call` field
1217
+ # must not be null once the call has genuinely started) by escaping run() raw.
1218
+ raise CallAborted(
1219
+ f"voice_call_translate_crashed: {type(exc).__name__}: {exc}",
1220
+ partial=self._timing_only_outcome(started_at),
1221
+ ) from exc
1222
+
1223
+ def _timing_only_outcome(self, started_at: datetime) -> CallOutcome:
1224
+ ended_at = datetime.now(timezone.utc)
1225
+ return CallOutcome(
1226
+ calls=(),
1227
+ turns=0,
1228
+ started_at=format_rfc3339_millis(started_at),
1229
+ ended_at=format_rfc3339_millis(ended_at),
1230
+ duration_ms=_duration_ms(started_at, ended_at),
1231
+ )
1232
+
1233
+ async def _translate_report(
1234
+ self,
1235
+ report: SimulationReport,
1236
+ *,
1237
+ runtime: EnvironmentRuntime,
1238
+ scenario_key: str,
1239
+ started_at: datetime,
1240
+ ) -> CallOutcome:
1241
+ case = report.test_cases[0] if report.test_cases else None
1242
+ case_started_at = (
1243
+ case.started_at
1244
+ if case is not None and case.started_at is not None
1245
+ else started_at
1246
+ )
1247
+ ended_at = (
1248
+ case.ended_at
1249
+ if case is not None and case.ended_at is not None
1250
+ else datetime.now(timezone.utc)
1251
+ )
1252
+ turns = (
1253
+ len(case.result.messages)
1254
+ if case is not None and case.result is not None
1255
+ else 0
1256
+ )
1257
+
1258
+ transcript_artifact: str | None = None
1259
+ recording_artifacts: list[str] = []
1260
+ # Evidence belongs to the call attempt, not only to successful calls. Collect it before
1261
+ # interpreting the simulator status so a timeout/agent failure still carries the exact
1262
+ # tool activity in its partial receipt. Previously the early CallAborted below discarded
1263
+ # every tool call from failed calls, making a real upstream tool error indistinguishable
1264
+ # from a proxy/transport failure.
1265
+ calls = self._collect_calls(runtime) if case is not None else ()
1266
+ # Provider-hosted agents execute tools outside the guest process, so their
1267
+ # authoritative call evidence is returned by Vapi/Retell after the call.
1268
+ # Preserve provider-native controls such as ``end_call`` because scenario
1269
+ # checks may verify termination ordering. Fall back to that observed stream
1270
+ # when the submitted backend exposes no local trace seam; never infer calls
1271
+ # from transcript prose.
1272
+ if not calls and case is not None:
1273
+ calls = _collect_provider_tool_calls(case)
1274
+ if calls:
1275
+ tool_trace = "\n".join(
1276
+ json.dumps(
1277
+ {
1278
+ "name": call.name,
1279
+ "arguments": call.arguments,
1280
+ "result": call.result,
1281
+ "ok": call.ok,
1282
+ "error": call.error,
1283
+ "refused": call.refused,
1284
+ "at": call.at,
1285
+ },
1286
+ sort_keys=True,
1287
+ default=str,
1288
+ )
1289
+ for call in calls
1290
+ ).encode("utf-8")
1291
+ await self._adapter.upload_artifact(
1292
+ tool_trace,
1293
+ kind=ArtifactKind.TOOL_TRACE,
1294
+ scenario_key=scenario_key,
1295
+ )
1296
+ if case is not None and case.result is not None:
1297
+ result = case.result
1298
+ if result.transcript:
1299
+ transcript_payload = json.dumps(
1300
+ {
1301
+ "schema_version": "futureagi.call-transcript.v1",
1302
+ "transcript": result.transcript,
1303
+ "messages": result.messages,
1304
+ },
1305
+ sort_keys=True,
1306
+ default=str,
1307
+ ).encode("utf-8")
1308
+ transcript_artifact = await self._adapter.upload_artifact(
1309
+ transcript_payload,
1310
+ kind=ArtifactKind.TRANSCRIPT,
1311
+ scenario_key=scenario_key,
1312
+ )
1313
+ for path_str, kind in (
1314
+ (result.audio_combined_path, ArtifactKind.RECORDING_COMBINED),
1315
+ (result.audio_stereo_path, ArtifactKind.RECORDING_STEREO),
1316
+ (result.audio_input_path, ArtifactKind.RECORDING_CUSTOMER),
1317
+ (result.audio_output_path, ArtifactKind.RECORDING_ASSISTANT),
1318
+ ):
1319
+ if not path_str:
1320
+ continue
1321
+ path = Path(path_str)
1322
+ if not path.is_file():
1323
+ continue
1324
+ artifact_id = await self._adapter.upload_artifact(
1325
+ path.read_bytes(),
1326
+ kind=kind,
1327
+ scenario_key=scenario_key,
1328
+ )
1329
+ if artifact_id is not None:
1330
+ recording_artifacts.append(artifact_id)
1331
+
1332
+ base = CallOutcome(
1333
+ calls=calls,
1334
+ turns=turns,
1335
+ started_at=format_rfc3339_millis(case_started_at),
1336
+ ended_at=format_rfc3339_millis(ended_at),
1337
+ duration_ms=_duration_ms(case_started_at, ended_at),
1338
+ transcript_artifact=transcript_artifact,
1339
+ recording_artifacts=tuple(recording_artifacts),
1340
+ messages=tuple(
1341
+ case.result.messages or ()
1342
+ if case is not None and case.result is not None
1343
+ else ()
1344
+ ),
1345
+ stop_reason=(
1346
+ str(case.result.metadata.get("stop_reason"))
1347
+ if case is not None
1348
+ and case.result is not None
1349
+ and case.result.metadata.get("stop_reason")
1350
+ else None
1351
+ ),
1352
+ )
1353
+
1354
+ if case is None:
1355
+ raise CallAborted(
1356
+ "voice_call_no_test_case: SimulationReport carried no test case",
1357
+ partial=base,
1358
+ )
1359
+
1360
+ if case.status is TestCaseStatus.AGENT_UNAVAILABLE:
1361
+ # world-handle-interface.md: "the agent never joined" is a WORLD failure, not a
1362
+ # scenario one -- the agent is part of the world, so the scheduler retires it and
1363
+ # retries elsewhere. Verified against the engine's own source (engines/livekit.py):
1364
+ # this status fires ONLY on a readiness-stage timeout with a session already started
1365
+ # but no target dispatched -- exactly "dispatch fails, agent never joins," never a
1366
+ # mid-call condition.
1367
+ reason = (
1368
+ case.failure.message
1369
+ if case.failure is not None
1370
+ else "agent_unavailable"
1371
+ )
1372
+ raise WorldUnavailable(f"target agent never joined the room: {reason}")
1373
+
1374
+ # A genuinely silent agent-first call (agent joined, zero conversational turns) reaches
1375
+ # the real engine (engines/livekit.py::_conversation_outcome) as FAILED with code
1376
+ # "no_conversation" or "conversation_silence_timeout" and zero messages -- never as a
1377
+ # COMPLETED case with zero turns (COMPLETED requires >= min_turn_messages AND role
1378
+ # alternation, so the engine cannot produce that shape). Scoped to zero turns only: a
1379
+ # short-but-nonzero conversation on either code still failed the completion bar for a real
1380
+ # reason and must stay a CallAborted below.
1381
+ is_silent_agent = (
1382
+ case.status is TestCaseStatus.FAILED
1383
+ and turns == 0
1384
+ and case.failure is not None
1385
+ and case.failure.code in _SILENT_AGENT_FAILURE_CODES
1386
+ )
1387
+
1388
+ # An intake agent may ask thirty to fifty questions, so a deadline is an ordinary outcome.
1389
+ ran_out_of_time = (
1390
+ case.status is TestCaseStatus.TIMED_OUT
1391
+ and turns >= _GRADEABLE_AFTER_TIMEOUT_TURNS
1392
+ )
1393
+
1394
+ if (
1395
+ case.status is not TestCaseStatus.COMPLETED
1396
+ and not is_silent_agent
1397
+ and not ran_out_of_time
1398
+ ):
1399
+ reason = (
1400
+ case.failure.message if case.failure is not None else case.status.value
1401
+ )
1402
+ attributed = _attributed_stall(case)
1403
+ if attributed is not None:
1404
+ code, reason = attributed
1405
+ raise CallAborted(reason, partial=base, code=code)
1406
+ if (
1407
+ case.failure is not None
1408
+ and case.failure.code == "target_agent_tool_failed"
1409
+ ):
1410
+ raise CallAborted(
1411
+ case.failure.message,
1412
+ partial=base,
1413
+ code="target_agent_tool_failed",
1414
+ )
1415
+ raise CallAborted(
1416
+ f"voice_call_not_completed: {case.status.value}: {reason}", partial=base
1417
+ )
1418
+
1419
+ # Never fabricate calls for a call that produced no conversation -- the scheduler's own
1420
+ # coverage guarantee turns an empty `calls` tuple into evidence_missing/simulator
1421
+ # regardless of turns (hosted_scheduler.py's own unconditioned-on-turns rule).
1422
+ calls = () if is_silent_agent else base.calls
1423
+ # Copied rather than rebuilt field by field: relisting them dropped `messages` silently,
1424
+ # and the judge then had no transcript to settle a spoken claim against.
1425
+ return replace(base, calls=calls)
1426
+
1427
+ def _collect_calls(self, runtime: EnvironmentRuntime) -> tuple[Call, ...]:
1428
+ seam = self._context.evidence_seam
1429
+ file_calls = _collect_file_tool_calls(runtime)
1430
+ if file_calls:
1431
+ return file_calls
1432
+ if seam is EvidenceSeam.HTTP_TOOL:
1433
+ return _collect_http_tool_calls(runtime)
1434
+ if seam is EvidenceSeam.TOOL_TRACE:
1435
+ return _collect_tool_trace_calls(runtime)
1436
+ # Unrecognized/None (should not happen for a `kind: process` bundle past preflight --
1437
+ # bundle_v2.py requires `evidence_seam` whenever `kind is PROCESS` -- but degrading rather
1438
+ # than crashing keeps this on the scheduler's own evidence_missing path, never a raw
1439
+ # exception).
1440
+ return ()