agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1340 @@
1
+ """FutureAGIResultSink — local write + real platform submission.
2
+
3
+ Composes ``LocalFilesystemResultSink`` and adds ``submit(...)`` that POSTs
4
+ report data to the Future AGI platform using the ALK ingestion endpoints:
5
+
6
+ POST /simulate/alk-simulate/run-tests/{run_test_id}/test-executions/
7
+ POST /simulate/alk-simulate/test-executions/{test_execution_id}/batch/
8
+ PATCH /simulate/alk-simulate/call-executions/{call_execution_id}/result/
9
+
10
+ Configuration is env-driven so local runs stay unaffected when the platform
11
+ target is not set. Submission accepts either the internal service bearer or
12
+ the external API-key pair:
13
+
14
+ FI_BASE_URL / FUTURE_AGI_API_URL / AGENT_LEARNING_API_URL — base URL
15
+ FI_API_KEY / FUTURE_AGI_API_KEY / AGENT_LEARNING_API_KEY — x-api-key
16
+ FI_SECRET_KEY / FUTURE_AGI_SECRET_KEY / AGENT_LEARNING_SECRET_KEY — x-secret-key
17
+ FI_RUN_TEST_ID / FUTURE_AGI_RUN_TEST_ID / AGENT_LEARNING_RUN_TEST_ID — target run test
18
+ FI_TEST_EXECUTION_ID / … — optional pre-created TestExecution (hosted runs); when
19
+ set the sink submits into it instead of creating one from the run test.
20
+
21
+ When any of those are absent the sink records ``status: "not_configured"``
22
+ in ``submission.json`` and returns cleanly — no HTTP is attempted.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import hashlib
28
+ import json
29
+ import logging
30
+ import os
31
+ from datetime import datetime, timezone
32
+ from pathlib import Path
33
+ from typing import Any
34
+
35
+ import httpx
36
+
37
+ from fi.simulate.runtime import (
38
+ CanonicalEvent,
39
+ SimulationPlan,
40
+ SimulationReport,
41
+ SimulationSpec,
42
+ )
43
+
44
+ from .filesystem import LocalFilesystemResultSink
45
+
46
+ logger = logging.getLogger("fi.simulate.results.futureagi")
47
+
48
+ _STATUS_MAP = {
49
+ "completed": "completed",
50
+ "failed": "failed",
51
+ "cancelled": "cancelled",
52
+ "timed_out": "failed",
53
+ "agent_unavailable": "failed",
54
+ }
55
+ _API_KEY_ENV = ("FI_API_KEY", "FUTURE_AGI_API_KEY", "AGENT_LEARNING_API_KEY")
56
+ _SECRET_KEY_ENV = (
57
+ "FI_SECRET_KEY",
58
+ "FUTURE_AGI_SECRET_KEY",
59
+ "AGENT_LEARNING_SECRET_KEY",
60
+ )
61
+ _INTERNAL_SECRET_ENV = (
62
+ "FI_INTERNAL_SUBMIT_SECRET",
63
+ "ALK_RUNNER_INTERNAL_SECRET",
64
+ "INTERNAL_API_SECRET",
65
+ )
66
+ _API_URL_ENV = ("FI_BASE_URL", "FUTURE_AGI_API_URL", "AGENT_LEARNING_API_URL")
67
+ _RUN_TEST_ID_ENV = (
68
+ "FI_RUN_TEST_ID",
69
+ "FUTURE_AGI_RUN_TEST_ID",
70
+ "AGENT_LEARNING_RUN_TEST_ID",
71
+ )
72
+ _TEST_EXECUTION_ID_ENV = (
73
+ "FI_TEST_EXECUTION_ID",
74
+ "FUTURE_AGI_TEST_EXECUTION_ID",
75
+ "AGENT_LEARNING_TEST_EXECUTION_ID",
76
+ )
77
+ _HTTP_TIMEOUT_SECONDS = 60.0
78
+ _RECORDING_UPLOAD_TIMEOUT_SECONDS = 300.0
79
+ _CONTENT_TYPE_BY_EXT = {
80
+ ".wav": "audio/wav",
81
+ ".mp3": "audio/mpeg",
82
+ ".ogg": "audio/ogg",
83
+ ".webm": "audio/webm",
84
+ ".m4a": "audio/mp4",
85
+ }
86
+
87
+
88
+ class FutureAGIResultSink:
89
+ """Local sink + platform submission over HTTP."""
90
+
91
+ def __init__(
92
+ self,
93
+ *,
94
+ root: str | Path = ".fagi/runs",
95
+ api_url: str | None = None,
96
+ api_key_env: tuple[str, ...] = _API_KEY_ENV,
97
+ secret_key_env: tuple[str, ...] = _SECRET_KEY_ENV,
98
+ run_test_id: str | None = None,
99
+ test_execution_id: str | None = None,
100
+ ) -> None:
101
+ self._local = LocalFilesystemResultSink(root=root)
102
+ self._api_url = api_url or _first_env(_API_URL_ENV)
103
+ self._api_key_env = api_key_env
104
+ self._secret_key_env = secret_key_env
105
+ self._run_test_id = run_test_id or _first_env(_RUN_TEST_ID_ENV)
106
+ self._test_execution_id = test_execution_id or _first_env(
107
+ _TEST_EXECUTION_ID_ENV
108
+ )
109
+ self._event_count = 0
110
+ self._spec: SimulationSpec | None = None
111
+ self._plan: SimulationPlan | None = None
112
+ # Streaming state (hosted runs only). ``begin_stream`` opens the client
113
+ # and allocates rows up front; ``submit_case`` PATCHes one row by index as
114
+ # its case finishes; ``finalize_stream`` reconciles + closes.
115
+ self._streaming = False
116
+ self._stream_client: httpx.Client | None = None
117
+ self._stream_call_ids: list[str] = []
118
+ self._streamed_indices: set[int] = set()
119
+ self._stream_failures: dict[int, dict[str, Any]] = {}
120
+
121
+ @property
122
+ def run_directory(self) -> Path | None:
123
+ return self._local.run_directory
124
+
125
+ def prepare(
126
+ self,
127
+ spec: SimulationSpec,
128
+ plan: SimulationPlan | None = None,
129
+ ) -> Path:
130
+ self._spec = spec
131
+ self._plan = plan
132
+ self._event_count = 0
133
+ return self._local.prepare(spec, plan)
134
+
135
+ def write_event(self, event: CanonicalEvent) -> None:
136
+ self._event_count += 1
137
+ self._local.write_event(event)
138
+
139
+ def write_report(self, report: SimulationReport) -> Path:
140
+ report_path = self._local.write_report(report)
141
+ # When streaming, cases were already PATCHed one-by-one as they finished;
142
+ # finalize only reconciles the stragglers and writes submission.json.
143
+ # Otherwise the whole report is submitted here in one batch (local/chat).
144
+ if self._streaming:
145
+ self.finalize_stream(report)
146
+ else:
147
+ self.submit(report)
148
+ return report_path
149
+
150
+ def begin_stream(
151
+ self,
152
+ spec: SimulationSpec,
153
+ plan: SimulationPlan | None = None,
154
+ ) -> bool:
155
+ """Open a run-scoped submission session for per-case streaming.
156
+
157
+ Hosted only: a pre-created ``test_execution_id`` is the signal. Local and
158
+ chat runs (no pre-created execution) return ``False`` and keep the
159
+ batch-at-end path untouched. On any setup error we also return ``False``
160
+ and fall back to that path — streaming setup must never break the run.
161
+ """
162
+ if not self._test_execution_id:
163
+ return False
164
+ api_key = _first_env(self._api_key_env)
165
+ secret_key = _first_env(self._secret_key_env)
166
+ internal_secret = _first_env(_INTERNAL_SECRET_ENV)
167
+ if _missing_config(
168
+ api_url=self._api_url,
169
+ api_key=api_key,
170
+ secret_key=secret_key,
171
+ internal_secret=internal_secret,
172
+ run_test_id=self._run_test_id,
173
+ ):
174
+ return False
175
+ try:
176
+ client = _open_client(self._api_url, api_key, secret_key, internal_secret)
177
+ call_ids = _allocate_call_ids(client, self._test_execution_id)
178
+ except Exception:
179
+ if self._stream_client is not None:
180
+ self._stream_client.close()
181
+ self._stream_client = None
182
+ return False
183
+
184
+ self._stream_client = client
185
+ self._stream_call_ids = call_ids
186
+ self._streamed_indices = set()
187
+ self._stream_failures = {}
188
+ self._streaming = True
189
+ return True
190
+
191
+ def submit_case(self, index: int, case: Any) -> None:
192
+ """PATCH one finished case into its pre-allocated CallExecution row.
193
+
194
+ Called off the event loop (``asyncio.to_thread``) from the runner's
195
+ per-case callback; ``httpx.Client`` is thread-safe, so the run-scoped
196
+ client is shared across concurrent case submissions. Failures are logged
197
+ and left for ``finalize_stream`` to reconcile — never raised.
198
+ """
199
+ if not self._streaming or self._stream_client is None:
200
+ return
201
+ if index >= len(self._stream_call_ids):
202
+ # More results than allocated rows: the platform under-provisioned.
203
+ # The batch path drops these silently via ``zip``; record it here so
204
+ # submission.json shows the drop.
205
+ self._stream_failures[index] = {
206
+ "index": index,
207
+ "reason": "no_allocated_row",
208
+ }
209
+ return
210
+ call_id = self._stream_call_ids[index]
211
+ try:
212
+ payload = _build_result_payload(case)
213
+ recording_url = _maybe_upload_recording(self._stream_client, call_id, case)
214
+ if recording_url:
215
+ payload["recording_url"] = recording_url
216
+ stereo_url = _maybe_upload_stereo_recording(
217
+ self._stream_client, call_id, case
218
+ )
219
+ if stereo_url:
220
+ payload["stereo_recording_url"] = stereo_url
221
+ _attach_channel_recordings(self._stream_client, call_id, case, payload)
222
+ _stamp_result_digest(payload)
223
+ resp = self._stream_client.patch(
224
+ f"/simulate/api/alk-simulate/call-executions/{call_id}/result/",
225
+ json=payload,
226
+ )
227
+ if resp.is_error:
228
+ self._stream_failures[index] = {
229
+ "index": index,
230
+ "call_execution_id": call_id,
231
+ "status_code": resp.status_code,
232
+ "body": _safe_body(resp),
233
+ }
234
+ logger.warning(
235
+ "case submission http error",
236
+ extra={
237
+ "case_index": index,
238
+ "call_execution_id": call_id,
239
+ "status_code": resp.status_code,
240
+ },
241
+ )
242
+ return
243
+ self._streamed_indices.add(index)
244
+ self._stream_failures.pop(index, None)
245
+ except Exception as exc:
246
+ self._stream_failures[index] = {
247
+ "index": index,
248
+ "call_execution_id": call_id,
249
+ "error": f"{type(exc).__name__}: {exc}",
250
+ }
251
+ logger.warning(
252
+ "case submission failed",
253
+ extra={
254
+ "case_index": index,
255
+ "call_execution_id": call_id,
256
+ "error": f"{type(exc).__name__}: {exc}",
257
+ },
258
+ )
259
+
260
+ def case_started(self, index: int) -> None:
261
+ """PATCH a pre-allocated CallExecution row to ONGOING the moment its case
262
+ starts, so the platform shows progress instead of PENDING → terminal.
263
+
264
+ Purely cosmetic and best-effort. Unlike ``submit_case`` this must NOT
265
+ record into ``_stream_failures`` / ``_streamed_indices`` — those drive
266
+ ``finalize_stream`` result reconciliation, and a missed status ping is not
267
+ a missed result. The backend gates the update on PENDING, so a lost, late,
268
+ or duplicate ping can never overwrite a terminal result; failures are
269
+ swallowed.
270
+ """
271
+ if not self._streaming or self._stream_client is None:
272
+ return
273
+ if index >= len(self._stream_call_ids):
274
+ return
275
+ call_id = self._stream_call_ids[index]
276
+ try:
277
+ self._stream_client.patch(
278
+ f"/simulate/api/alk-simulate/call-executions/{call_id}/status/",
279
+ json={"status": "ongoing"},
280
+ )
281
+ except Exception:
282
+ # Non-authoritative: never let a status ping disturb the run or the
283
+ # reconciliation bookkeeping.
284
+ pass
285
+
286
+ def finalize_stream(self, report: SimulationReport) -> dict[str, Any]:
287
+ """Reconcile any case the stream missed, then close the session.
288
+
289
+ Runs after the engine returns, so no cases are in flight. Any index not
290
+ already streamed (PATCH failure, or a whole-report failure that never fired
291
+ callbacks) is retried here. ``submission.json`` records ``status:
292
+ submitted`` when at least one case landed and ``failed`` when none did —
293
+ ``child_entrypoint`` gates job success on that field.
294
+ """
295
+ run_directory = self._local.run_directory
296
+ for index, case in enumerate(report.test_cases):
297
+ if index in self._streamed_indices:
298
+ continue
299
+ self.submit_case(index, case)
300
+
301
+ submitted = [self._stream_call_ids[i] for i in sorted(self._streamed_indices)]
302
+ failed = [detail for _, detail in sorted(self._stream_failures.items())]
303
+ # Every allocated row failing to submit is a failed submission — not a
304
+ # green job with zero results landed (the batch path signalled this by
305
+ # letting the exception propagate to ``submit``). Partial failures stay
306
+ # "submitted": the cases that did land are real.
307
+ all_failed = (
308
+ bool(report.test_cases)
309
+ and bool(self._stream_call_ids)
310
+ and not self._streamed_indices
311
+ )
312
+ submission: dict[str, Any] = {
313
+ "schema_version": "futureagi.submission.v1",
314
+ "run_id": report.run_id,
315
+ "report_hash": report.report_hash,
316
+ "test_cases": len(report.test_cases),
317
+ "events_recorded": self._event_count,
318
+ "api_url": self._api_url,
319
+ "run_test_id": self._run_test_id,
320
+ "test_execution_id": self._test_execution_id,
321
+ "streamed": True,
322
+ "status": "failed" if all_failed else "submitted",
323
+ "allocated_call_executions": list(self._stream_call_ids),
324
+ "submitted_call_executions": submitted,
325
+ "failed_call_executions": failed,
326
+ "generated_at": datetime.now(timezone.utc).isoformat(),
327
+ }
328
+ if all_failed:
329
+ submission["reason"] = "stream_all_cases_failed"
330
+ if self._stream_client is not None:
331
+ self._stream_client.close()
332
+ self._stream_client = None
333
+ self._streaming = False
334
+ if run_directory is not None:
335
+ _write_submission(run_directory, submission)
336
+ return submission
337
+
338
+ def submit(self, report: SimulationReport) -> dict[str, Any]:
339
+ run_directory = self._local.run_directory
340
+ if run_directory is None:
341
+ raise RuntimeError("result_sink_not_prepared")
342
+
343
+ api_key = _first_env(self._api_key_env)
344
+ secret_key = _first_env(self._secret_key_env)
345
+ internal_secret = _first_env(_INTERNAL_SECRET_ENV)
346
+
347
+ submission: dict[str, Any] = {
348
+ "schema_version": "futureagi.submission.v1",
349
+ "run_id": report.run_id,
350
+ "report_hash": report.report_hash,
351
+ "test_cases": len(report.test_cases),
352
+ "artifact_count": len(report.artifacts.entries),
353
+ "events_recorded": self._event_count,
354
+ "api_url": self._api_url,
355
+ "run_test_id": self._run_test_id,
356
+ "test_execution_id": self._test_execution_id,
357
+ "generated_at": datetime.now(timezone.utc).isoformat(),
358
+ }
359
+
360
+ missing = _missing_config(
361
+ api_url=self._api_url,
362
+ api_key=api_key,
363
+ secret_key=secret_key,
364
+ internal_secret=internal_secret,
365
+ run_test_id=self._run_test_id,
366
+ )
367
+ if missing:
368
+ submission["status"] = "not_configured"
369
+ submission["reason"] = "missing_config: " + ",".join(missing)
370
+ logger.warning(
371
+ "submission not configured", extra={"missing": ",".join(missing)}
372
+ )
373
+ _write_submission(run_directory, submission)
374
+ return submission
375
+
376
+ try:
377
+ outcome = _submit_via_http(
378
+ report=report,
379
+ base_url=self._api_url,
380
+ api_key=api_key,
381
+ secret_key=secret_key,
382
+ internal_secret=internal_secret,
383
+ run_test_id=self._run_test_id,
384
+ test_execution_id=self._test_execution_id,
385
+ )
386
+ submission.update(outcome)
387
+ submission["status"] = "submitted"
388
+ logger.info(
389
+ "submission ok",
390
+ extra={
391
+ "run_test_id": self._run_test_id,
392
+ "test_execution_id": self._test_execution_id,
393
+ },
394
+ )
395
+ except Exception as exc:
396
+ submission["status"] = "failed"
397
+ submission["reason"] = f"submission_error: {exc.__class__.__name__}: {exc}"
398
+ logger.error(
399
+ "submission failed",
400
+ extra={
401
+ "run_test_id": self._run_test_id,
402
+ "test_execution_id": self._test_execution_id,
403
+ "error": f"{exc.__class__.__name__}: {exc}",
404
+ },
405
+ )
406
+
407
+ _write_submission(run_directory, submission)
408
+ return submission
409
+
410
+
411
+ def _first_env(names: tuple[str, ...]) -> str | None:
412
+ for name in names:
413
+ value = os.environ.get(name)
414
+ if value:
415
+ return value
416
+ return None
417
+
418
+
419
+ def _missing_config(
420
+ *,
421
+ api_url: str | None,
422
+ api_key: str | None,
423
+ secret_key: str | None,
424
+ internal_secret: str | None,
425
+ run_test_id: str | None,
426
+ ) -> list[str]:
427
+ missing: list[str] = []
428
+ if not api_url:
429
+ missing.append("api_url")
430
+ if not internal_secret:
431
+ if not api_key:
432
+ missing.append("api_key")
433
+ if not secret_key:
434
+ missing.append("secret_key")
435
+ if not run_test_id:
436
+ missing.append("run_test_id")
437
+ return missing
438
+
439
+
440
+ def _open_client(
441
+ base_url: str,
442
+ api_key: str | None,
443
+ secret_key: str | None,
444
+ internal_secret: str | None = None,
445
+ ) -> httpx.Client:
446
+ """Build the ALK ingestion HTTP client (shared by batch + streaming paths).
447
+
448
+ No client-level Content-Type: httpx sets application/json for ``json=`` calls
449
+ and multipart/form-data (with boundary) for the ``files=`` recording upload.
450
+ A fixed application/json here silently breaks the multipart upload.
451
+ """
452
+ headers: dict[str, str] = {}
453
+ if api_key and secret_key:
454
+ headers.update({"x-api-key": api_key, "x-secret-key": secret_key})
455
+ # These are alternative authentication modes. A developer/hosted-runner
456
+ # environment can legitimately contain both sets of variables; sending
457
+ # both lets the backend's bearer authenticator win and loses the customer
458
+ # organization carried by the API-key pair. Prefer the explicit customer
459
+ # identity whenever it is complete, and use the internal token only as the
460
+ # service-to-service fallback.
461
+ elif internal_secret:
462
+ headers["Authorization"] = f"Bearer {internal_secret}"
463
+ return httpx.Client(
464
+ base_url=base_url.rstrip("/"),
465
+ headers=headers,
466
+ timeout=_HTTP_TIMEOUT_SECONDS,
467
+ )
468
+
469
+
470
+ def _ensure_test_execution(client: httpx.Client, run_test_id: str) -> str:
471
+ """Create a TestExecution from the run test (local runs only)."""
472
+ start = client.post(
473
+ f"/simulate/api/alk-simulate/run-tests/{run_test_id}/test-executions/",
474
+ json={},
475
+ )
476
+ start.raise_for_status()
477
+ return _unwrap(start.json())["test_execution_id"]
478
+
479
+
480
+ def _allocate_call_ids(
481
+ client: httpx.Client,
482
+ test_execution_id: str,
483
+ count: int | None = None,
484
+ ) -> list[str]:
485
+ """Claim exact local-report rows, or all pre-created hosted rows."""
486
+ if count is not None and count <= 0:
487
+ return []
488
+
489
+ call_execution_ids: list[str] = []
490
+ for _ in range(64): # hard cap to prevent runaway
491
+ remaining = count - len(call_execution_ids) if count is not None else None
492
+ resp = client.post(
493
+ f"/simulate/api/alk-simulate/test-executions/{test_execution_id}/batch/",
494
+ json={"count": remaining} if remaining is not None else {},
495
+ )
496
+ resp.raise_for_status()
497
+ body = _unwrap(resp.json())
498
+ allocated = body["call_execution_ids"]
499
+ if remaining is not None and len(allocated) > remaining:
500
+ raise RuntimeError("backend allocated more call executions than requested")
501
+ call_execution_ids.extend(allocated)
502
+ if count is not None and len(call_execution_ids) == count:
503
+ break
504
+ if not allocated or not body.get("has_more"):
505
+ if count is None:
506
+ break
507
+ raise RuntimeError(
508
+ f"backend allocated {len(call_execution_ids)} of {count} required call executions"
509
+ )
510
+
511
+ if count is not None and len(call_execution_ids) != count:
512
+ raise RuntimeError(
513
+ f"backend allocated {len(call_execution_ids)} of {count} required call executions"
514
+ )
515
+ return call_execution_ids
516
+
517
+
518
+ def _submit_via_http(
519
+ *,
520
+ report: SimulationReport,
521
+ base_url: str,
522
+ api_key: str | None,
523
+ secret_key: str | None,
524
+ run_test_id: str,
525
+ internal_secret: str | None = None,
526
+ test_execution_id: str | None = None,
527
+ ) -> dict[str, Any]:
528
+ with _open_client(base_url, api_key, secret_key, internal_secret) as client:
529
+ # Hosted runs submit into a TestExecution the platform pre-created; local
530
+ # runs create one here from the run test.
531
+ if not test_execution_id:
532
+ test_execution_id = _ensure_test_execution(client, run_test_id)
533
+
534
+ call_execution_ids = _allocate_call_ids(
535
+ client,
536
+ test_execution_id,
537
+ len(report.test_cases),
538
+ )
539
+
540
+ submitted_ids: list[str] = []
541
+ failed: list[dict[str, Any]] = []
542
+ for call_id, case in zip(call_execution_ids, report.test_cases):
543
+ payload = _build_result_payload(case)
544
+ recording_url = _maybe_upload_recording(client, call_id, case)
545
+ if recording_url:
546
+ payload["recording_url"] = recording_url
547
+ stereo_url = _maybe_upload_stereo_recording(client, call_id, case)
548
+ if stereo_url:
549
+ payload["stereo_recording_url"] = stereo_url
550
+ _attach_channel_recordings(client, call_id, case, payload)
551
+ _stamp_result_digest(payload)
552
+ resp = client.patch(
553
+ f"/simulate/api/alk-simulate/call-executions/{call_id}/result/",
554
+ json=payload,
555
+ )
556
+ if resp.is_error:
557
+ failed.append(
558
+ {
559
+ "call_execution_id": call_id,
560
+ "status_code": resp.status_code,
561
+ "body": _safe_body(resp),
562
+ }
563
+ )
564
+ else:
565
+ submitted_ids.append(call_id)
566
+
567
+ return {
568
+ "test_execution_id": test_execution_id,
569
+ "allocated_call_executions": call_execution_ids,
570
+ "submitted_call_executions": submitted_ids,
571
+ "failed_call_executions": failed,
572
+ }
573
+
574
+
575
+ def _unwrap(body: Any) -> dict[str, Any]:
576
+ if isinstance(body, dict) and "result" in body and isinstance(body["result"], dict):
577
+ return body["result"]
578
+ if isinstance(body, dict):
579
+ return body
580
+ raise ValueError(f"unexpected_response_shape: {body!r}")
581
+
582
+
583
+ def _safe_body(response: httpx.Response) -> Any:
584
+ try:
585
+ return response.json()
586
+ except Exception:
587
+ return response.text[:500]
588
+
589
+
590
+ def _build_result_payload(case) -> dict[str, Any]:
591
+ """Map a SimulationTestCaseResult into the ALK ingestion PATCH body.
592
+
593
+ Backend derives conversation metrics and CSAT from the transcript, so
594
+ the SDK only ships what it directly observed.
595
+ """
596
+ payload: dict[str, Any] = {
597
+ "status": _STATUS_MAP.get(case.status.value, "failed"),
598
+ }
599
+
600
+ started_at = case.started_at
601
+ ended_at = case.ended_at
602
+ # Case-level timestamps are unset for LiveKit runs (the engine does not
603
+ # stamp them). Fall back to the observed speech timing carried on each
604
+ # message so duration/start-time populate on the platform.
605
+ if started_at is None or ended_at is None:
606
+ speech_start, speech_end = _speech_bounds(case)
607
+ started_at = started_at or speech_start
608
+ ended_at = ended_at or speech_end
609
+
610
+ if started_at is not None:
611
+ payload["started_at"] = started_at.isoformat()
612
+ if ended_at is not None:
613
+ payload["ended_at"] = ended_at.isoformat()
614
+ if started_at is not None and ended_at is not None:
615
+ payload["duration_seconds"] = max(
616
+ int((ended_at - started_at).total_seconds()), 0
617
+ )
618
+
619
+ if case.failure is not None:
620
+ payload["ended_reason"] = case.failure.code
621
+ payload["error_message"] = case.failure.message or ""
622
+
623
+ result = case.result
624
+ transcript_segments: list[dict[str, Any]] = []
625
+ if result is not None:
626
+ transcript_segments = _extract_transcript_segments(result)
627
+ if transcript_segments:
628
+ payload["transcript"] = transcript_segments
629
+ if "ended_reason" not in payload:
630
+ stop_reason = result.metadata.get("stop_reason")
631
+ if isinstance(stop_reason, str) and stop_reason:
632
+ payload["ended_reason"] = stop_reason
633
+
634
+ recording_uri = _extract_recording_uri(result)
635
+ if recording_uri:
636
+ payload["recording_url"] = recording_uri
637
+
638
+ provider_call_data: dict[str, Any] = {}
639
+ existing_pcd = result.metadata.get("provider_call_data")
640
+ if isinstance(existing_pcd, dict):
641
+ provider_call_data = dict(existing_pcd)
642
+
643
+ # Fold the target agent's provider-reported usage/cost (captured by the
644
+ # SDK evidence layer — Vapi costBreakdown, Retell call_cost, LiveKit
645
+ # usage) into provider_call_data under the normalized ``usage.llm``
646
+ # shape the platform already reads for native voice. This is the
647
+ # agent-under-test's real usage — not the FutureAGI simulator's.
648
+ target = _target_provider_usage(case)
649
+ if target is not None:
650
+ provider_bucket = dict(provider_call_data.get(target.provider) or {})
651
+ if target.usage:
652
+ provider_bucket["usage"] = {
653
+ **(provider_bucket.get("usage") or {}),
654
+ "llm": target.usage,
655
+ }
656
+ if target.raw:
657
+ provider_bucket.setdefault("costBreakdown", target.raw)
658
+ if provider_bucket:
659
+ provider_call_data[target.provider] = provider_bucket
660
+ if target.cost_cents is not None:
661
+ payload["costs"] = {"cost_cents": target.cost_cents}
662
+
663
+ # Every LiveKit-engine run carries a truthy ``livekit`` marker so the
664
+ # platform's SpeakerRoleResolver detects the provider as LiveKit — its
665
+ # role map is direction-independent and already matches the SDK's
666
+ # tested-agent-perspective transcript. Without this a black-box target
667
+ # (no usage evidence) leaves ``provider_call_data`` empty, the platform
668
+ # falls back to VAPI, and an inbound default swaps agent/customer labels.
669
+ # A falsy ``{}`` is not enough — ``detect_provider`` treats it as absent.
670
+ if str(
671
+ result.metadata.get("engine")
672
+ ) == "livekit" and not provider_call_data.get("livekit"):
673
+ provider_call_data["livekit"] = {"engine": "livekit"}
674
+
675
+ if provider_call_data:
676
+ payload["provider_call_data"] = provider_call_data
677
+
678
+ summary = result.metadata.get("call_summary") or result.metadata.get("summary")
679
+ if isinstance(summary, str) and summary:
680
+ payload["call_summary"] = summary
681
+
682
+ call_metadata = {
683
+ k: v
684
+ for k, v in result.metadata.items()
685
+ if k
686
+ not in {
687
+ "provider_call_data",
688
+ "call_summary",
689
+ "summary",
690
+ "failure",
691
+ "status",
692
+ "test_case_id",
693
+ "run_id",
694
+ }
695
+ }
696
+ if call_metadata:
697
+ payload["call_metadata"] = _json_safe(call_metadata)
698
+
699
+ return payload
700
+
701
+
702
+ def _stamp_result_digest(payload: dict[str, Any]) -> None:
703
+ """Bind an idempotency key to exactly the canonical payload being submitted."""
704
+ payload.pop("result_digest", None)
705
+ payload["result_digest"] = (
706
+ "sha256:"
707
+ + hashlib.sha256(
708
+ json.dumps(
709
+ payload,
710
+ sort_keys=True,
711
+ separators=(",", ":"),
712
+ default=str,
713
+ ).encode()
714
+ ).hexdigest()
715
+ )
716
+
717
+
718
+ def _extract_transcript_segments(result) -> list[dict[str, Any]]:
719
+ """Convert TestCaseResult.messages into ALK transcript segments.
720
+
721
+ LiveKit engine emits each message with ``started_speaking_at`` and
722
+ ``stopped_speaking_at`` (seconds since epoch, from ``ChatMessage.metrics``).
723
+ We convert to millisecond offsets relative to the first speech timestamp
724
+ so ``ConversationMetricsCalculator`` can compute overlap-based interruption
725
+ counts, WPM and talk-ratio on the backend.
726
+ """
727
+ segments: list[dict[str, Any]] = []
728
+ typed_messages = [msg for msg in result.messages if isinstance(msg, dict)]
729
+
730
+ anchor = _first_speech_anchor(typed_messages)
731
+ for msg in typed_messages:
732
+ role = msg.get("role")
733
+ content = msg.get("content")
734
+ tool_calls = _normalize_tool_calls(msg.get("tool_calls"))
735
+ start_ms, end_ms = _resolve_message_timing_ms(msg, anchor)
736
+ latency_ms = _message_latency_ms(msg)
737
+
738
+ if role == "assistant":
739
+ # An assistant turn can carry text, tool calls, or both. Tool-call
740
+ # turns usually have empty content — emit them anyway as a
741
+ # ``tool_calls`` segment so the agent's real tool activity survives
742
+ # ingestion instead of being dropped by the empty-content guard.
743
+ if isinstance(content, str) and content:
744
+ segments.append(
745
+ _segment("assistant", content, start_ms, end_ms, latency_ms)
746
+ )
747
+ if tool_calls:
748
+ segments.append(
749
+ _segment(
750
+ "tool_calls",
751
+ _render_tool_calls(tool_calls),
752
+ start_ms,
753
+ end_ms,
754
+ latency_ms,
755
+ tool_calls=tool_calls,
756
+ )
757
+ )
758
+ continue
759
+
760
+ if role == "tool":
761
+ if isinstance(content, str) and content:
762
+ segments.append(
763
+ _segment(
764
+ "tool_call_result",
765
+ content,
766
+ start_ms,
767
+ end_ms,
768
+ None,
769
+ tool_call_id=msg.get("tool_call_id") or msg.get("id"),
770
+ )
771
+ )
772
+ continue
773
+
774
+ if not isinstance(content, str) or not content:
775
+ continue
776
+ if role in {"user", "customer"}:
777
+ speaker_role = "user"
778
+ elif role == "system":
779
+ speaker_role = "system"
780
+ else:
781
+ speaker_role = "unknown"
782
+ segments.append(_segment(speaker_role, content, start_ms, end_ms, None))
783
+
784
+ if segments:
785
+ return segments
786
+
787
+ if not result.transcript:
788
+ return []
789
+ for line in result.transcript.splitlines():
790
+ if ":" not in line:
791
+ continue
792
+ role_label, content = line.split(":", 1)
793
+ role_label = role_label.strip().lower()
794
+ content = content.strip()
795
+ if not content:
796
+ continue
797
+ if role_label in {"assistant", "agent", "bot"}:
798
+ speaker_role = "assistant"
799
+ elif role_label in {"customer", "user", "simulator", "caller"}:
800
+ speaker_role = "user"
801
+ else:
802
+ speaker_role = "unknown"
803
+ segments.append(
804
+ {
805
+ "speaker_role": speaker_role,
806
+ "content": content,
807
+ "start_time_ms": 0,
808
+ "end_time_ms": 0,
809
+ }
810
+ )
811
+ return segments
812
+
813
+
814
+ def _segment(
815
+ speaker_role: str,
816
+ content: str,
817
+ start_ms: int,
818
+ end_ms: int,
819
+ latency_ms: int | None,
820
+ *,
821
+ tool_calls: list[dict[str, Any]] | None = None,
822
+ tool_call_id: str | None = None,
823
+ ) -> dict[str, Any]:
824
+ seg: dict[str, Any] = {
825
+ "speaker_role": speaker_role,
826
+ "content": content,
827
+ "start_time_ms": start_ms,
828
+ "end_time_ms": end_ms,
829
+ }
830
+ if latency_ms is not None:
831
+ seg["latency_ms"] = latency_ms
832
+ if tool_calls:
833
+ seg["tool_calls"] = tool_calls
834
+ if tool_call_id:
835
+ seg["tool_call_id"] = tool_call_id
836
+ return seg
837
+
838
+
839
+ def _normalize_tool_calls(raw: Any) -> list[dict[str, Any]] | None:
840
+ """Coerce a message's tool_calls into a stable [{id, name, arguments}] shape.
841
+
842
+ Accepts both the flat SDK shape (``{"name", "arguments", "id"}``) and the
843
+ OpenAI/LiteLLM nested shape (``{"function": {"name", "arguments"}}``);
844
+ ``arguments`` is JSON-decoded when the provider ships it as a string.
845
+ """
846
+ if not raw or not isinstance(raw, (list, tuple)):
847
+ return None
848
+ calls: list[dict[str, Any]] = []
849
+ for tc in raw:
850
+ if not isinstance(tc, dict):
851
+ continue
852
+ fn = tc.get("function") if isinstance(tc.get("function"), dict) else {}
853
+ name = tc.get("name") or fn.get("name")
854
+ if not name:
855
+ continue
856
+ arguments = tc.get("arguments")
857
+ if arguments is None:
858
+ arguments = fn.get("arguments")
859
+ if isinstance(arguments, str):
860
+ try:
861
+ arguments = json.loads(arguments) if arguments.strip() else {}
862
+ except (ValueError, TypeError):
863
+ pass
864
+ calls.append(
865
+ {
866
+ "id": tc.get("id") or name,
867
+ "name": name,
868
+ "arguments": arguments if arguments is not None else {},
869
+ }
870
+ )
871
+ return calls or None
872
+
873
+
874
+ def _render_tool_calls(calls: list[dict[str, Any]]) -> str:
875
+ lines: list[str] = []
876
+ for call in calls:
877
+ args = call.get("arguments")
878
+ try:
879
+ rendered = json.dumps(args, ensure_ascii=False, sort_keys=True)
880
+ except (TypeError, ValueError):
881
+ rendered = str(args)
882
+ lines.append(f"{call['name']}({rendered})")
883
+ return "\n".join(lines)
884
+
885
+
886
+ def _message_latency_ms(msg: dict[str, Any]) -> int | None:
887
+ for key in ("latency_ms", "latency"):
888
+ value = msg.get(key)
889
+ if isinstance(value, (int, float)) and value > 0:
890
+ return int(value)
891
+ metrics = msg.get("metrics")
892
+ if isinstance(metrics, dict):
893
+ for key in ("latency_ms", "latency"):
894
+ value = metrics.get(key)
895
+ if isinstance(value, (int, float)) and value > 0:
896
+ return int(value)
897
+ return None
898
+
899
+
900
+ def _maybe_upload_recording(
901
+ client: httpx.Client, call_execution_id: str, case
902
+ ) -> str | None:
903
+ """Upload the case's audio file (if any) via a multipart POST.
904
+
905
+ Prefers a combined/mixed WAV, falls back to output-only then input-only.
906
+ Skips silently when no on-disk audio exists (e.g. ``record_audio=False``
907
+ on the runner, or the SDK already surfaced an HTTPS URL via
908
+ ``result.artifacts``). Returns the persisted ``recording_url`` to attach
909
+ to the ingestion PATCH, or None.
910
+ """
911
+ if case.result is None:
912
+ return None
913
+ audio_path = _select_audio_path(case.result)
914
+ if audio_path is None:
915
+ return None
916
+
917
+ filename = audio_path.name
918
+ content_type = _CONTENT_TYPE_BY_EXT.get(
919
+ audio_path.suffix.lower(), "application/octet-stream"
920
+ )
921
+ with audio_path.open("rb") as fh:
922
+ files = {"file": (filename, fh, content_type)}
923
+ data = {
924
+ "filename": filename,
925
+ "sha256": _sha256_file(audio_path),
926
+ "kind": _recording_kind(case.result, audio_path),
927
+ }
928
+ resp = client.post(
929
+ f"/simulate/api/alk-simulate/call-executions/{call_execution_id}/recording/",
930
+ files=files,
931
+ data=data,
932
+ timeout=_RECORDING_UPLOAD_TIMEOUT_SECONDS,
933
+ )
934
+ if resp.is_error:
935
+ return None
936
+ body = _unwrap(resp.json())
937
+ return body.get("recording_url")
938
+
939
+
940
+ def _maybe_upload_stereo_recording(
941
+ client: httpx.Client, call_execution_id: str, case
942
+ ) -> str | None:
943
+ """Upload the case's 2-channel stereo WAV (ch0 customer, ch1 assistant).
944
+
945
+ Uses the same multipart endpoint as ``_maybe_upload_recording`` and returns
946
+ the persisted URL for ``stereo_recording_url``, or None when absent.
947
+ """
948
+ if case.result is None:
949
+ return None
950
+ stereo_path = getattr(case.result, "audio_stereo_path", None)
951
+ if not stereo_path:
952
+ return None
953
+ path = Path(str(stereo_path)).expanduser()
954
+ if not (path.exists() and path.is_file() and path.stat().st_size > 0):
955
+ return None
956
+
957
+ filename = path.name
958
+ content_type = _CONTENT_TYPE_BY_EXT.get(
959
+ path.suffix.lower(), "application/octet-stream"
960
+ )
961
+ with path.open("rb") as fh:
962
+ files = {"file": (filename, fh, content_type)}
963
+ data = {
964
+ "filename": filename,
965
+ "sha256": _sha256_file(path),
966
+ "kind": "stereo",
967
+ }
968
+ resp = client.post(
969
+ f"/simulate/api/alk-simulate/call-executions/{call_execution_id}/recording/",
970
+ files=files,
971
+ data=data,
972
+ timeout=_RECORDING_UPLOAD_TIMEOUT_SECONDS,
973
+ )
974
+ if resp.is_error:
975
+ return None
976
+ body = _unwrap(resp.json())
977
+ return body.get("recording_url")
978
+
979
+
980
+ def _upload_audio_file(
981
+ client: httpx.Client, call_execution_id: str, path: Path, *, kind: str
982
+ ) -> str | None:
983
+ """POST a single on-disk WAV to the recording endpoint; return its URL."""
984
+ filename = path.name
985
+ content_type = _CONTENT_TYPE_BY_EXT.get(
986
+ path.suffix.lower(), "application/octet-stream"
987
+ )
988
+ with path.open("rb") as fh:
989
+ files = {"file": (filename, fh, content_type)}
990
+ data = {"filename": filename, "sha256": _sha256_file(path), "kind": kind}
991
+ resp = client.post(
992
+ f"/simulate/api/alk-simulate/call-executions/{call_execution_id}/recording/",
993
+ files=files,
994
+ data=data,
995
+ timeout=_RECORDING_UPLOAD_TIMEOUT_SECONDS,
996
+ )
997
+ if resp.is_error:
998
+ return None
999
+ return _unwrap(resp.json()).get("recording_url")
1000
+
1001
+
1002
+ def _sha256_file(path: Path) -> str:
1003
+ digest = hashlib.sha256()
1004
+ with path.open("rb") as stream:
1005
+ while chunk := stream.read(1024 * 1024):
1006
+ digest.update(chunk)
1007
+ return digest.hexdigest()
1008
+
1009
+
1010
+ def _upload_channel_recording(
1011
+ client: httpx.Client, call_execution_id: str, case, attr: str
1012
+ ) -> str | None:
1013
+ """Upload one per-speaker mono WAV named by ``attr`` on the case result."""
1014
+ if case.result is None:
1015
+ return None
1016
+ value = getattr(case.result, attr, None)
1017
+ if not value:
1018
+ return None
1019
+ path = Path(str(value)).expanduser()
1020
+ if not (path.exists() and path.is_file() and path.stat().st_size > 0):
1021
+ return None
1022
+ kind = "assistant" if attr == "audio_output_path" else "customer"
1023
+ return _upload_audio_file(client, call_execution_id, path, kind=kind)
1024
+
1025
+
1026
+ def _attach_channel_recordings(
1027
+ client: httpx.Client, call_execution_id: str, case, payload: dict[str, Any]
1028
+ ) -> None:
1029
+ """Upload the per-speaker assistant/customer mono tracks and fold their URLs
1030
+ into ``provider_call_data.livekit.recording`` so evals mapped to
1031
+ ``call.assistant_recording`` / ``call.customer_recording`` resolve. LiveKit
1032
+ runs otherwise only produce combined + stereo, leaving the per-channel
1033
+ variables empty. ``audio_output_path`` is the target/assistant track,
1034
+ ``audio_input_path`` is the simulator/customer track (matching the stereo
1035
+ channel order ch0 customer, ch1 assistant)."""
1036
+ assistant_url = _upload_channel_recording(
1037
+ client, call_execution_id, case, "audio_output_path"
1038
+ )
1039
+ customer_url = _upload_channel_recording(
1040
+ client, call_execution_id, case, "audio_input_path"
1041
+ )
1042
+ recording: dict[str, str] = {}
1043
+ if assistant_url:
1044
+ recording["assistant"] = assistant_url
1045
+ if customer_url:
1046
+ recording["customer"] = customer_url
1047
+ if not recording:
1048
+ return
1049
+ provider_call_data = payload.setdefault("provider_call_data", {})
1050
+ livekit = provider_call_data.setdefault("livekit", {})
1051
+ if not isinstance(livekit, dict):
1052
+ return
1053
+ livekit["recording"] = {**(livekit.get("recording") or {}), **recording}
1054
+
1055
+
1056
+ def _select_audio_path(result) -> Path | None:
1057
+ for candidate in (
1058
+ result.audio_combined_path,
1059
+ result.audio_output_path,
1060
+ result.audio_input_path,
1061
+ ):
1062
+ if not candidate:
1063
+ continue
1064
+ path = Path(str(candidate)).expanduser()
1065
+ if path.exists() and path.is_file() and path.stat().st_size > 0:
1066
+ return path
1067
+ return None
1068
+
1069
+
1070
+ def _recording_kind(result, path: Path) -> str:
1071
+ if (
1072
+ result.audio_combined_path
1073
+ and path == Path(str(result.audio_combined_path)).expanduser()
1074
+ ):
1075
+ return "combined"
1076
+ if (
1077
+ result.audio_output_path
1078
+ and path == Path(str(result.audio_output_path)).expanduser()
1079
+ ):
1080
+ return "assistant"
1081
+ return "customer"
1082
+
1083
+
1084
+ _TARGET_PROVIDERS = ("vapi", "retell", "livekit")
1085
+
1086
+
1087
+ class _TargetUsage:
1088
+ """Normalized target-agent usage extracted from one provider's evidence."""
1089
+
1090
+ __slots__ = ("provider", "usage", "cost_cents", "raw")
1091
+
1092
+ def __init__(
1093
+ self,
1094
+ provider: str,
1095
+ usage: dict[str, int] | None,
1096
+ cost_cents: int | None,
1097
+ raw: dict[str, Any] | None,
1098
+ ) -> None:
1099
+ self.provider = provider
1100
+ self.usage = usage
1101
+ self.cost_cents = cost_cents
1102
+ self.raw = raw
1103
+
1104
+
1105
+ def _target_provider_usage(case) -> _TargetUsage | None:
1106
+ """Pull the target agent's provider-reported usage from case evidence.
1107
+
1108
+ Provider-agnostic: dispatches to a per-provider extractor because each
1109
+ provider reports cost/tokens in a different shape (Vapi costBreakdown,
1110
+ Retell call_cost + llm_token_usage, LiveKit normalized usage). Returns a
1111
+ ``_TargetUsage`` with a normalized ``usage`` (``prompt_tokens`` /
1112
+ ``completion_tokens`` / ``total_tokens``) and ``cost_cents``, or None when
1113
+ no target evidence surfaced usage (e.g. a black-box self-hosted target).
1114
+ """
1115
+ evidence = getattr(case, "evidence", None) or []
1116
+ for source in evidence:
1117
+ metadata = getattr(source, "metadata", None) or {}
1118
+ provider = metadata.get("provider")
1119
+ if provider not in _TARGET_PROVIDERS:
1120
+ continue
1121
+ extractor = _PROVIDER_USAGE_EXTRACTORS.get(provider)
1122
+ if extractor is None:
1123
+ continue
1124
+ result = extractor(metadata)
1125
+ if result is not None:
1126
+ return result
1127
+ return None
1128
+
1129
+
1130
+ def _vapi_usage(metadata: dict[str, Any]) -> _TargetUsage | None:
1131
+ cost = metadata.get("cost") if isinstance(metadata.get("cost"), dict) else {}
1132
+ breakdown = (
1133
+ cost.get("breakdown") if isinstance(cost.get("breakdown"), dict) else None
1134
+ )
1135
+ usage = None
1136
+ if breakdown:
1137
+ prompt = breakdown.get("llmPromptTokens", breakdown.get("promptTokens"))
1138
+ completion = breakdown.get(
1139
+ "llmCompletionTokens", breakdown.get("completionTokens")
1140
+ )
1141
+ usage = _normalized_usage(prompt, completion)
1142
+ cost_cents = _dollars_to_cents(cost.get("total"))
1143
+ if usage is None and cost_cents is None:
1144
+ return None
1145
+ return _TargetUsage("vapi", usage, cost_cents, breakdown)
1146
+
1147
+
1148
+ def _retell_usage(metadata: dict[str, Any]) -> _TargetUsage | None:
1149
+ token_usage = metadata.get("usage")
1150
+ usage = None
1151
+ if isinstance(token_usage, dict):
1152
+ # Retell may report prompt/completion directly, or per-request `values`
1153
+ # (total tokens only, no split).
1154
+ prompt = token_usage.get("num_input_tokens", token_usage.get("prompt_tokens"))
1155
+ completion = token_usage.get(
1156
+ "num_output_tokens", token_usage.get("completion_tokens")
1157
+ )
1158
+ if prompt is not None or completion is not None:
1159
+ usage = _normalized_usage(prompt, completion)
1160
+ else:
1161
+ values = token_usage.get("values")
1162
+ if isinstance(values, list) and values:
1163
+ total = sum(_coerce_int(v) for v in values)
1164
+ if total:
1165
+ usage = {"total_tokens": total}
1166
+ call_cost = metadata.get("cost") if isinstance(metadata.get("cost"), dict) else {}
1167
+ # Retell reports combined_cost already in cents.
1168
+ cost_cents = _coerce_int_or_none(call_cost.get("combined_cost"))
1169
+ if usage is None and cost_cents is None:
1170
+ return None
1171
+ return _TargetUsage("retell", usage, cost_cents, call_cost or None)
1172
+
1173
+
1174
+ def _livekit_usage(metadata: dict[str, Any]) -> _TargetUsage | None:
1175
+ # A LiveKit target that reports a normalized usage blob back through the
1176
+ # evidence layer (self-hosted worker). Absent for black-box targets.
1177
+ usage_blob = metadata.get("usage")
1178
+ if not isinstance(usage_blob, dict):
1179
+ return None
1180
+ llm = (
1181
+ usage_blob.get("llm") if isinstance(usage_blob.get("llm"), dict) else usage_blob
1182
+ )
1183
+ prompt = llm.get("prompt_tokens", llm.get("promptTokens"))
1184
+ completion = llm.get("completion_tokens", llm.get("completionTokens"))
1185
+ usage = _normalized_usage(prompt, completion)
1186
+ cost_cents = _dollars_to_cents(
1187
+ (metadata.get("cost") or {}).get("total")
1188
+ if isinstance(metadata.get("cost"), dict)
1189
+ else None
1190
+ )
1191
+ if usage is None and cost_cents is None:
1192
+ return None
1193
+ return _TargetUsage("livekit", usage, cost_cents, None)
1194
+
1195
+
1196
+ _PROVIDER_USAGE_EXTRACTORS = {
1197
+ "vapi": _vapi_usage,
1198
+ "retell": _retell_usage,
1199
+ "livekit": _livekit_usage,
1200
+ }
1201
+
1202
+
1203
+ def _normalized_usage(prompt: Any, completion: Any) -> dict[str, int] | None:
1204
+ if prompt is None and completion is None:
1205
+ return None
1206
+ prompt_i = _coerce_int(prompt)
1207
+ completion_i = _coerce_int(completion)
1208
+ return {
1209
+ "prompt_tokens": prompt_i,
1210
+ "completion_tokens": completion_i,
1211
+ "total_tokens": prompt_i + completion_i,
1212
+ }
1213
+
1214
+
1215
+ def _dollars_to_cents(value: Any) -> int | None:
1216
+ dollars = _coerce_float(value)
1217
+ return int(round(dollars * 100)) if dollars is not None else None
1218
+
1219
+
1220
+ def _coerce_int(value: Any) -> int:
1221
+ try:
1222
+ return int(value or 0)
1223
+ except (TypeError, ValueError):
1224
+ return 0
1225
+
1226
+
1227
+ def _coerce_int_or_none(value: Any) -> int | None:
1228
+ if value is None:
1229
+ return None
1230
+ try:
1231
+ return int(round(float(value)))
1232
+ except (TypeError, ValueError):
1233
+ return None
1234
+
1235
+
1236
+ def _coerce_float(value: Any) -> float | None:
1237
+ try:
1238
+ return float(value) if value is not None else None
1239
+ except (TypeError, ValueError):
1240
+ return None
1241
+
1242
+
1243
+ def _speech_bounds(case) -> tuple[Any, Any]:
1244
+ """Return (start, end) datetimes from a case's message speech timing.
1245
+
1246
+ LiveKit messages carry ``started_speaking_at`` / ``stopped_speaking_at``
1247
+ as epoch seconds; the earliest start and latest stop bound the actual
1248
+ conversation. Returns (None, None) when no timing is available.
1249
+ """
1250
+ if case.result is None:
1251
+ return None, None
1252
+ starts: list[float] = []
1253
+ ends: list[float] = []
1254
+ for msg in case.result.messages:
1255
+ if not isinstance(msg, dict):
1256
+ continue
1257
+ start = msg.get("started_speaking_at") or msg.get("created_at")
1258
+ stop = msg.get("stopped_speaking_at") or msg.get("created_at")
1259
+ if isinstance(start, (int, float)) and start > 0:
1260
+ starts.append(float(start))
1261
+ if isinstance(stop, (int, float)) and stop > 0:
1262
+ ends.append(float(stop))
1263
+ if not starts or not ends:
1264
+ return None, None
1265
+ start_dt = datetime.fromtimestamp(min(starts), tz=timezone.utc)
1266
+ end_dt = datetime.fromtimestamp(max(ends), tz=timezone.utc)
1267
+ if end_dt < start_dt:
1268
+ end_dt = start_dt
1269
+ return start_dt, end_dt
1270
+
1271
+
1272
+ def _first_speech_anchor(messages: list[dict[str, Any]]) -> float | None:
1273
+ for msg in messages:
1274
+ for key in ("started_speaking_at", "created_at"):
1275
+ value = msg.get(key)
1276
+ if isinstance(value, (int, float)) and value > 0:
1277
+ return float(value)
1278
+ return None
1279
+
1280
+
1281
+ def _resolve_message_timing_ms(
1282
+ msg: dict[str, Any], anchor: float | None
1283
+ ) -> tuple[int, int]:
1284
+ """Return (start_ms, end_ms) relative to the first-speech anchor.
1285
+
1286
+ Falls back to ``created_at`` when speech-timing metrics are missing (text
1287
+ turns, providers that don't report the metric). Zero is used as the last
1288
+ resort — the backend metrics calculator degrades gracefully when timings
1289
+ collapse to zero-duration.
1290
+ """
1291
+ if anchor is None:
1292
+ return 0, 0
1293
+
1294
+ start_raw = msg.get("started_speaking_at") or msg.get("created_at") or 0.0
1295
+ stop_raw = (
1296
+ msg.get("stopped_speaking_at") or msg.get("created_at") or start_raw or 0.0
1297
+ )
1298
+ start_ms = (
1299
+ max(int(round((float(start_raw) - anchor) * 1000)), 0) if start_raw else 0
1300
+ )
1301
+ end_ms = (
1302
+ max(int(round((float(stop_raw) - anchor) * 1000)), start_ms)
1303
+ if stop_raw
1304
+ else start_ms
1305
+ )
1306
+ return start_ms, end_ms
1307
+
1308
+
1309
+ def _extract_recording_uri(result) -> str | None:
1310
+ for artifact in result.artifacts:
1311
+ artifact_type = getattr(artifact, "type", None)
1312
+ if artifact_type == "audio" and getattr(artifact, "uri", None):
1313
+ return artifact.uri
1314
+ for candidate in (
1315
+ result.audio_combined_path,
1316
+ result.audio_output_path,
1317
+ result.audio_input_path,
1318
+ ):
1319
+ if candidate and str(candidate).startswith(("http://", "https://")):
1320
+ return str(candidate)
1321
+ return None
1322
+
1323
+
1324
+ def _json_safe(value: Any) -> Any:
1325
+ try:
1326
+ json.dumps(value)
1327
+ return value
1328
+ except (TypeError, ValueError):
1329
+ return json.loads(json.dumps(value, default=str))
1330
+
1331
+
1332
+ def _write_submission(run_directory: Path, payload: dict[str, Any]) -> None:
1333
+ submission_path = run_directory / "submission.json"
1334
+ submission_path.write_text(
1335
+ json.dumps(payload, indent=2, sort_keys=True, default=str) + "\n",
1336
+ encoding="utf-8",
1337
+ )
1338
+
1339
+
1340
+ __all__ = ["FutureAGIResultSink"]