agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1209 @@
1
+ """The one thing the harness hands a suite to.
2
+
3
+ The harness does not run scenarios. It builds a world, writes scenarios against it, and calls
4
+ `simulate` once. Everything after that belongs to ALK: how many run at a time, whether the person
5
+ is typed to or phoned, where the audio goes, what a report looks like.
6
+
7
+ That split matters more than it looks. While the harness ran scenarios itself, one at a time,
8
+ through its own conversation loop, a suite was only as good as the harness's patience: a run took
9
+ as many turns of the chat as it had scenarios, and the simulator driving it was not the one the
10
+ product ships. Handing over means the suite runs the same way whether a person triggered it from
11
+ the UI, a script did, or nobody did.
12
+
13
+ Chat and voice are one path here. A contract-only chat spec may run as an in-process target; a
14
+ repository-backed chat agent runs its submitted service and is reached through its declared HTTP
15
+ or WebSocket ingress; a voice agent is reached through its declared realtime transport. All three
16
+ receive the same isolated world, setup, checks and report. Only the target adapter differs, and a
17
+ repository-backed agent is never reconstructed from its extracted prompt.
18
+
19
+ A run is a folder. One simulation over a suite is one run, kept whole, so a session accumulates
20
+ runs that can be compared rather than one result file that the next run overwrites.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import asyncio
26
+ import inspect
27
+ import json
28
+ import logging
29
+ import re
30
+ import time
31
+ from collections.abc import Callable
32
+ from dataclasses import asdict
33
+ from datetime import UTC, datetime
34
+ from pathlib import Path
35
+ from typing import TYPE_CHECKING, Any
36
+
37
+ from ..contract import AgentContract
38
+ from ..scenario import Scenario
39
+ from ..world.runtime import Call
40
+ from .grade import Judgement, Result
41
+
42
+ if TYPE_CHECKING:
43
+ from .conversation import Exchange
44
+
45
+ RUNS = "runs"
46
+ RUN = "run.json"
47
+ RESULT = "result.json"
48
+ TRANSCRIPT = "transcript.txt"
49
+ CALLS = "calls.json"
50
+ logger = logging.getLogger(__name__)
51
+
52
+ # How many scenarios run at once by default. One, because the shipped default should be the one
53
+ # that cannot surprise anybody: a voice suite places real calls that cost real money, and fanning
54
+ # out to twenty is a bad thing to learn from a bill.
55
+ CONCURRENCY = 1
56
+
57
+ # What ALK calls the world and the person, per modality. Both are registry names it validates
58
+ # against the plugin's own manifest, so a typo is an error here rather than a confusing run.
59
+ WORLDS = {"text": ("chat", "chat"), "voice": ("voice", "voice")}
60
+ SIMULATORS = {"text": "synthetic_user", "voice": "livekit_simulator"}
61
+
62
+
63
+ def spoken_to(contract: AgentContract) -> bool:
64
+ """Whether this agent is spoken to rather than typed to."""
65
+ return (contract.modality or "text").strip().lower() == "voice"
66
+
67
+
68
+ def new_run_id() -> str:
69
+ return datetime.now(UTC).strftime("run-%Y%m%d-%H%M%S")
70
+
71
+
72
+ def run_root(destination: Path, run_id: str) -> Path:
73
+ return Path(destination) / RUNS / run_id
74
+
75
+
76
+ def every_run(destination: Path) -> list[dict[str, Any]]:
77
+ """Every run in this session, newest first, finished or not.
78
+
79
+ A run that is still going is reported too, from the results already written. `run.json` is
80
+ written once, at the end, so requiring it meant an hour-long suite showed nothing at all
81
+ while its results sat on disk: the scenario that finished forty minutes ago was as invisible
82
+ as the one that had not started. `finished` says which kind each is.
83
+ """
84
+ root = Path(destination) / RUNS
85
+ if not root.exists():
86
+ return []
87
+ found: list[dict[str, Any]] = []
88
+ for folder in sorted(root.iterdir(), reverse=True):
89
+ if not folder.is_dir():
90
+ continue
91
+ kept = folder / RUN
92
+ if kept.exists():
93
+ try:
94
+ summary = json.loads(kept.read_text(encoding="utf-8"))
95
+ except Exception: # noqa: BLE001 - one unreadable run never hides the rest
96
+ continue
97
+ summary["finished"] = True
98
+ found.append(summary)
99
+ continue
100
+ done = _cases_so_far(folder)
101
+ if done:
102
+ found.append(
103
+ {
104
+ "run_id": folder.name,
105
+ "finished": False,
106
+ "scenarios": len(done),
107
+ "passed": sum(1 for one in done if one.get("passed")),
108
+ "seconds": round(sum(one.get("seconds") or 0 for one in done), 1),
109
+ "results": done,
110
+ }
111
+ )
112
+ return found
113
+
114
+
115
+ def _cases_so_far(folder: Path) -> list[dict[str, Any]]:
116
+ """The scenarios of an unfinished run that have already been written."""
117
+ done: list[dict[str, Any]] = []
118
+ for case in sorted(folder.iterdir()):
119
+ kept = case / RESULT
120
+ if not case.is_dir() or not kept.exists():
121
+ continue
122
+ try:
123
+ one = json.loads(kept.read_text(encoding="utf-8"))
124
+ except Exception: # noqa: BLE001 - a result being written this instant is not an error
125
+ continue
126
+ done.append(
127
+ {
128
+ "scenario": one.get("scenario", case.name),
129
+ "passed": bool(one.get("passed")),
130
+ "met": one.get("met"),
131
+ "of": len(one.get("checkpoints") or []),
132
+ "seconds": one.get("seconds"),
133
+ "recording": one.get("recording", ""),
134
+ "problems": one.get("problems") or [],
135
+ }
136
+ )
137
+ return done
138
+
139
+
140
+ def read_run(destination: Path, run_id: str) -> dict[str, Any]:
141
+ """One run in full: its summary, and every scenario's result, transcript and calls.
142
+
143
+ Read from the folder rather than held in memory, so the harness can be asked about a run
144
+ that happened before it was started, and about any single call inside one.
145
+ """
146
+ root = run_root(destination, run_id)
147
+ kept = root / RUN
148
+ if not root.exists():
149
+ raise FileNotFoundError(f"no run {run_id} in {destination}")
150
+ # A run still going has no summary yet, but the scenarios it has finished are readable and
151
+ # worth reading. Only a folder that is not there at all is an error.
152
+ summary = (
153
+ json.loads(kept.read_text(encoding="utf-8"))
154
+ if kept.exists()
155
+ else {"run_id": run_id, "finished": False, "passed": 0}
156
+ )
157
+ summary.setdefault("finished", kept.exists())
158
+ scenarios: list[dict[str, Any]] = []
159
+ for folder in sorted(root.iterdir()):
160
+ if not folder.is_dir() or not (folder / RESULT).exists():
161
+ continue
162
+ one = json.loads((folder / RESULT).read_text(encoding="utf-8"))
163
+ one["transcript"] = _text(folder / TRANSCRIPT)
164
+ one["calls_detail"] = _json(folder / CALLS)
165
+ scenarios.append(one)
166
+ summary["scenarios"] = scenarios
167
+ return summary
168
+
169
+
170
+ def _text(path: Path) -> str:
171
+ return path.read_text(encoding="utf-8") if path.exists() else ""
172
+
173
+
174
+ def _json(path: Path) -> list[dict[str, Any]]:
175
+ return json.loads(path.read_text(encoding="utf-8")) if path.exists() else []
176
+
177
+
178
+ async def simulate(
179
+ scenarios: list[Scenario],
180
+ contract: AgentContract,
181
+ world_root: Path,
182
+ *,
183
+ destination: Path | None = None,
184
+ model: str | None = None,
185
+ concurrency: int = CONCURRENCY,
186
+ run_id: str = "",
187
+ on_case_start: Callable[[Scenario], Any] | None = None,
188
+ on_case_done: Callable[[Result], Any] | None = None,
189
+ on_exchange: Callable[[str, dict[str, Any]], Any] | None = None,
190
+ ) -> dict[str, Any]:
191
+ """Run a whole suite through ALK and write it out as one run.
192
+
193
+ Returns the run's summary. Results are in the order they were asked for, not the order they
194
+ finished, so a report reads the same however it was scheduled.
195
+ """
196
+ from .models import for_roles
197
+
198
+ destination = Path(destination or world_root)
199
+ if (Path(world_root) / "environment.json").exists() and concurrency != 1:
200
+ # Restored source worlds point at the submitted Compose project's one real datastore.
201
+ # Until a provisioner can clone that entire project per case, parallel cases would reset
202
+ # and mutate the same database underneath each other. Serialize explicitly rather than
203
+ # offering fast but invalid isolation.
204
+ logger.warning(
205
+ "source-provisioned scenarios share one isolated Compose project; forcing "
206
+ "concurrency from %s to 1",
207
+ concurrency,
208
+ )
209
+ concurrency = 1
210
+ run_id = run_id or new_run_id()
211
+ root = run_root(destination, run_id)
212
+ root.mkdir(parents=True, exist_ok=True)
213
+ roles = for_roles(model)
214
+
215
+ # Durations must not jump when the host clock is corrected (common on laptops/VMs). Keep the
216
+ # human timestamp separately and measure elapsed time with the monotonic clock.
217
+ started_at = datetime.now(UTC).isoformat(timespec="seconds")
218
+ started = time.monotonic()
219
+ room = asyncio.Semaphore(max(1, concurrency))
220
+ ordered: list[Result | None] = [None] * len(scenarios)
221
+
222
+ async def one(index: int, scenario: Scenario) -> None:
223
+ async with room:
224
+ if on_case_start:
225
+ notified = on_case_start(scenario)
226
+ if inspect.isawaitable(notified):
227
+ await notified
228
+ began = time.monotonic()
229
+ folder = root / scenario.name
230
+ folder.mkdir(parents=True, exist_ok=True)
231
+ try:
232
+ result = await _run_one(
233
+ scenario,
234
+ contract,
235
+ world_root,
236
+ folder,
237
+ roles=roles,
238
+ on_exchange=(
239
+ (lambda turn: on_exchange(scenario.name, turn))
240
+ if on_exchange
241
+ else None
242
+ ),
243
+ )
244
+ except Exception as failed: # noqa: BLE001 - one bad scenario never stops the suite
245
+ result = Result(
246
+ scenario=scenario.name,
247
+ tests=scenario.tests,
248
+ problems=[f"{type(failed).__name__}: {failed}"],
249
+ # This is a terminal outcome for the attempted scenario, but it is not an
250
+ # agent result. Keeping an explicit ending prevents downstream artifact
251
+ # readers from confusing an exception-shaped partial record with a call
252
+ # that is still in progress.
253
+ ended="failed",
254
+ )
255
+ result.seconds = round(time.monotonic() - began, 1)
256
+ _write_case(folder, result)
257
+ ordered[index] = result
258
+ if on_case_done:
259
+ notified = on_case_done(result)
260
+ if inspect.isawaitable(notified):
261
+ await notified
262
+
263
+ await asyncio.gather(
264
+ *(one(index, scenario) for index, scenario in enumerate(scenarios))
265
+ )
266
+ results = [one for one in ordered if one is not None]
267
+
268
+ summary = {
269
+ "run_id": run_id,
270
+ "agent": contract.agent,
271
+ "modality": contract.modality or "text",
272
+ "started": started_at,
273
+ "seconds": round(time.monotonic() - started, 1),
274
+ "concurrency": concurrency,
275
+ "models": roles,
276
+ "scenarios": len(results),
277
+ "passed": sum(1 for one in results if one.passed),
278
+ # A scenario that could not be executed is an infrastructure/harness outcome, not a
279
+ # weak-agent grade. The CLI and hosted worker use this count to keep those two result
280
+ # classes distinct all the way to the platform.
281
+ "unrunnable": sum(1 for one in results if one.problems),
282
+ "spent_usd": round(sum(one.spent_usd for one in results), 4),
283
+ # Averaged across the scenarios that reported them, so a suite has one line per metric
284
+ # rather than a number nobody compares. Only over the runs that actually measured it:
285
+ # averaging a missing metric as zero would make a suite look worse the more of it failed
286
+ # to run, which is the opposite of informative.
287
+ "metrics": _averaged([one.measured for one in results]),
288
+ "results": [
289
+ {
290
+ "scenario": one.scenario,
291
+ "passed": one.passed,
292
+ "met": one.met,
293
+ "of": len(one.checkpoints),
294
+ "seconds": one.seconds,
295
+ "recording": one.recording,
296
+ "problems": one.problems,
297
+ }
298
+ for one in results
299
+ ],
300
+ }
301
+ (root / RUN).write_text(
302
+ json.dumps(summary, indent=2, default=str), encoding="utf-8"
303
+ )
304
+ return summary
305
+
306
+
307
+ def _write_case(folder: Path, result: Result) -> None:
308
+ """One scenario's result, transcript and calls, each in the form it is read in.
309
+
310
+ The transcript is written as text because it is read by people, and the calls as JSON
311
+ because they are read by the UI and by the harness looking into a single call.
312
+ """
313
+ body = asdict(result)
314
+ body["passed"] = result.passed
315
+ body["met"] = result.met
316
+ detail = body.pop("calls_detail", None) or []
317
+ (folder / RESULT).write_text(
318
+ json.dumps(body, indent=2, default=str), encoding="utf-8"
319
+ )
320
+ (folder / TRANSCRIPT).write_text(result.transcript or "", encoding="utf-8")
321
+ (folder / CALLS).write_text(
322
+ json.dumps(detail, indent=2, default=str), encoding="utf-8"
323
+ )
324
+
325
+
326
+ async def _run_one(
327
+ scenario: Scenario,
328
+ contract: AgentContract,
329
+ world_root: Path,
330
+ folder: Path,
331
+ *,
332
+ roles: dict[str, str],
333
+ on_exchange: Callable[[dict[str, Any]], Any] | None = None,
334
+ ) -> Result:
335
+ """One scenario, in its own world, through ALK's runner.
336
+
337
+ The world is prepared here and handed in, rather than named in the spec, because isolation
338
+ is ours to guarantee: every scenario starts from the same frozen base with only its own
339
+ setup applied, and a world shared between cases would let the first one decide what the
340
+ second is graded against.
341
+ """
342
+
343
+ from ..folder import apply_setup, check_ready
344
+ from ..world.snapshot import restore
345
+
346
+ spoken = spoken_to(contract)
347
+ kind = "voice" if spoken else "text"
348
+ adapter, world_kind = WORLDS[kind]
349
+
350
+ world = restore(world_root)
351
+ try:
352
+ world.reset()
353
+ applied = apply_setup(scenario, world)
354
+ if not applied.ok:
355
+ raise RuntimeError(f"the scenario's setup did not run: {applied.said}")
356
+ ready = check_ready(scenario, world)
357
+ if not ready.ok:
358
+ raise RuntimeError(
359
+ f"the world is not ready for this scenario: {ready.said}. Running it would "
360
+ "test us rather than the agent."
361
+ )
362
+ # The setup's own calls are not the agent's.
363
+ world.calls = []
364
+
365
+ if not spoken:
366
+ # Typed, and driven by a model rather than by ALK's chat simulator.
367
+ #
368
+ # That simulator is deterministic on purpose: an untyped persona gets three fixed
369
+ # lines ("Can you give me the exact next step…"), and a typed one renders utterances
370
+ # from a compiled behaviour policy. Reproducible, and not a simulation of a person.
371
+ # A suite whose user says the same three things to every agent tests one path and
372
+ # calls it coverage.
373
+ #
374
+ # So the conversation is driven here, by a model reading the simulator prompt the
375
+ # build stage wrote for this agent. Everything around it is unchanged: same world,
376
+ # same setup, same checks, same run folder.
377
+ return await _typed_to(
378
+ scenario, contract, world, world_root, folder, roles=roles
379
+ )
380
+
381
+ # Spoken. The agent is not here: it runs in Vapi, with its own prompt, its own model
382
+ # and its own voice, and the only thing that changes is where its tools are answered.
383
+ # ALK places the call and drives a simulated caller that is a real model over STT and
384
+ # TTS, so this half was never deterministic.
385
+ return await _spoken_to(
386
+ scenario,
387
+ contract,
388
+ world,
389
+ world_root,
390
+ folder,
391
+ roles=roles,
392
+ on_exchange=on_exchange,
393
+ )
394
+ finally:
395
+ try:
396
+ world.close()
397
+ except Exception: # cleanup must never replace a completed scenario result
398
+ logger.exception("world cleanup failed after scenario %s", scenario.name)
399
+
400
+
401
+ def _found_audio(directory: Path) -> Path | None:
402
+ """The recording a run left behind, if it left one.
403
+
404
+ Asked of the directory rather than taken on trust from whatever placed the call: a runner
405
+ that exits badly still returns a path, and a path is not a file.
406
+ """
407
+ if not directory.exists():
408
+ return None
409
+ for path in sorted(directory.rglob("*")):
410
+ if path.is_file() and path.suffix.lower() in (".wav", ".mp3", ".ogg", ".m4a"):
411
+ return path
412
+ return None
413
+
414
+
415
+ async def _typed_to(
416
+ scenario: Scenario,
417
+ contract: AgentContract,
418
+ world: Any,
419
+ world_root: Path,
420
+ folder: Path,
421
+ *,
422
+ roles: dict[str, str],
423
+ ) -> Result:
424
+ """A typed conversation, with a model on both sides.
425
+
426
+ The same grading as every other run: the world it is handed is already set up, and what it
427
+ leaves behind is what the checks read.
428
+ """
429
+ from ..catalogue import load_catalogue
430
+ from . import converse
431
+ from .grade import (
432
+ checkpoints,
433
+ grade_sub_goals,
434
+ judge,
435
+ judge_suite_evals,
436
+ reconcile_task_completion,
437
+ ungraded_sub_goals,
438
+ )
439
+ from .targets import resolve
440
+
441
+ repository_backed = bool(
442
+ contract.runtime or contract.tool_entrypoints or contract.implementation
443
+ )
444
+ if repository_backed:
445
+ agent = resolve("repository")(
446
+ contract,
447
+ world,
448
+ world_root=world_root,
449
+ trace_path=folder / "agent-tool-calls.jsonl",
450
+ scenario_name=scenario.name,
451
+ )
452
+ else:
453
+ agent = resolve("local")(contract, world, model=roles["agent"])
454
+ transcript = await converse(
455
+ agent, scenario, contract, world_root=world_root, model=roles["user"]
456
+ )
457
+ catalogue = load_catalogue(world_root)
458
+ settled = grade_sub_goals(world, scenario, catalogue, transcript.calls)
459
+ ending = ", ".join(
460
+ f"{name}: {len(rows)} rows"
461
+ for name, rows in sorted(world.observe().state.items())
462
+ )
463
+ judgements, judged_cost = await judge(
464
+ scenario, transcript, contract, catalogue, model=roles["judge"], ending=ending
465
+ )
466
+ suite_judgements = judge_suite_evals(
467
+ catalogue.suite_evals, scenario, transcript, contract, ending=ending
468
+ )
469
+ judgements += reconcile_task_completion(suite_judgements, settled, judgements)
470
+ result = Result(
471
+ scenario=scenario.name,
472
+ tests=scenario.tests,
473
+ problems=[
474
+ f"{name} is not in this catalogue, so nothing graded it"
475
+ for name in ungraded_sub_goals(scenario, catalogue)
476
+ ],
477
+ state_failures=[f"{one.name}: {one.said}" for one in settled if not one.held],
478
+ conduct=judgements,
479
+ checkpoints=checkpoints(settled, judgements),
480
+ crashes=[f"{call.name}: {call.error}" for call in transcript.crashed()],
481
+ ended=transcript.ended,
482
+ turns=len(transcript.exchanges),
483
+ calls=len(transcript.calls),
484
+ spent_usd=transcript.spent_usd + judged_cost,
485
+ transcript=transcript.spoken(),
486
+ exchanges=[
487
+ {"speaker": turn.speaker, "text": turn.text}
488
+ for turn in transcript.exchanges
489
+ ],
490
+ actions=transcript.actions(),
491
+ )
492
+ result.calls_detail = _calls_of(transcript.calls)
493
+ return result
494
+
495
+
496
+ def _calls_of(calls: Any) -> list[dict[str, Any]]:
497
+ """Every call in full, for the timeline and for anyone asking what one call did."""
498
+ return [
499
+ {
500
+ "name": call.name,
501
+ "arguments": call.arguments,
502
+ "result": str(call.result)[:2000],
503
+ "ok": call.ok,
504
+ "refused": call.refused,
505
+ "error": call.error,
506
+ "at": getattr(call, "at", 0.0),
507
+ }
508
+ for call in calls
509
+ ]
510
+
511
+
512
+ def _said(line: str) -> Exchange:
513
+ """One transcript line as a turn, with its role read off rather than left in the text.
514
+
515
+ The line arrives already labelled ("assistant: ..."). Keeping that label in the text made the
516
+ judge read ``agent: assistant: ...``, two speakers deep for every turn.
517
+ """
518
+ from .conversation import Exchange
519
+
520
+ role, _, text = line.partition(":")
521
+ named = role.strip().lower()
522
+ if named in ("assistant", "agent"):
523
+ return Exchange("agent", text.strip())
524
+ if named in ("user", "customer"):
525
+ return Exchange("customer", text.strip())
526
+ return Exchange("customer", line.strip())
527
+
528
+
529
+ async def _spoken_to(
530
+ scenario: Scenario,
531
+ contract: AgentContract,
532
+ world: Any,
533
+ world_root: Path,
534
+ folder: Path,
535
+ *,
536
+ roles: dict[str, str],
537
+ on_exchange: Callable[[dict[str, Any]], Any] | None = None,
538
+ ) -> Result:
539
+ """A real call, with the agent's own tools answered by this world.
540
+
541
+ The agent under test is not reconstructed here and is not running in this process. It is the
542
+ hosted assistant, with its own prompt, model and voice; the only thing that changes for the
543
+ duration is where its tool calls are sent. That makes this the more faithful of the two
544
+ paths, and the reason a spoken suite is worth more than a typed one.
545
+
546
+ The call itself belongs to ALK, which drives a simulated caller through speech: a real model
547
+ behind STT and TTS, not a script.
548
+ """
549
+ import os
550
+ import time
551
+
552
+ from ..catalogue import load_catalogue
553
+ from .call import place_the_call
554
+ from .conversation import Transcript
555
+ from .evidence import measured, newest_report, spoken_times, tracks_in
556
+ from .grade import (
557
+ checkpoints,
558
+ grade_sub_goals,
559
+ judge,
560
+ judge_suite_evals,
561
+ reconcile_task_completion,
562
+ ungraded_sub_goals,
563
+ )
564
+ from .live import wire
565
+ from .tools import configure_source_voice, missing_prerequisites
566
+
567
+ configure_source_voice(world_root, contract)
568
+ stopping = missing_prerequisites(world_root, contract)
569
+ if stopping:
570
+ raise RuntimeError("cannot place a call:\n - " + "\n - ".join(stopping))
571
+
572
+ loop = asyncio.get_running_loop()
573
+
574
+ def live_exchange(turn: dict[str, Any]) -> None:
575
+ if on_exchange:
576
+ normalized = _normalize_live_exchange(turn)
577
+ loop.call_soon_threadsafe(on_exchange, normalized)
578
+
579
+ def placed_once() -> tuple[int, dict[str, Any], str]:
580
+ """Everything about the call, off the event loop.
581
+
582
+ Wiring reads a subprocess's stdout and the call itself blocks for minutes. Run inline
583
+ they freeze whatever loop is hosting this, which for the UI means the stream, the status
584
+ endpoint and the stop button all stop with it.
585
+ """
586
+ _world, instruction, webhook, tunnel, _url, _moved = wire(
587
+ scenario,
588
+ world_root,
589
+ world=world,
590
+ trace_path=folder / "agent-tool-calls.jsonl",
591
+ )
592
+ started = time.time()
593
+ runtime_output = ""
594
+ try:
595
+ sdk_output = folder / "sdk"
596
+ os.environ["HARNESS_VOICE_OUTPUT_ROOT"] = str(sdk_output.resolve())
597
+ os.environ["HARNESS_INSTRUCTION"] = instruction
598
+ os.environ["HARNESS_SCENARIO"] = scenario.name
599
+ # The caller is never handed the grader's pass question: `tests` is written about
600
+ # the agent in the third person, so as an objective it reads as a rubric rather
601
+ # than a motive. What this person wants is already in the instruction.
602
+ os.environ.pop("HARNESS_OUTCOME", None)
603
+ os.environ["HARNESS_PERSONA"] = json.dumps(
604
+ scenario.persona.model_dump(exclude_none=True)
605
+ if scenario.persona is not None
606
+ else {"name": "customer"}
607
+ )
608
+ os.environ["HARNESS_INITIAL_MESSAGE"] = (
609
+ scenario.persona.initial_message if scenario.persona is not None else ""
610
+ )
611
+ os.environ["HARNESS_SCRIPTED_CALLER"] = json.dumps(
612
+ scenario.persona.scripted_caller
613
+ if scenario.persona is not None
614
+ and scenario.persona.scripted_caller is not None
615
+ else {}
616
+ )
617
+ os.environ["HARNESS_FIXTURE"] = json.dumps(
618
+ scenario.fixture, ensure_ascii=False, default=str
619
+ )
620
+ # A scenario that asks to be heard through background noise selects a clip for the
621
+ # caller's environment; the voice engine mixes it under the caller. Cleared otherwise so
622
+ # a previous call's noise never leaks into a quiet one.
623
+ from ..background_noise import scenario_source
624
+
625
+ source = scenario_source(
626
+ getattr(scenario, "background_noise", False),
627
+ scenario.fixture,
628
+ seed=scenario.name,
629
+ )
630
+ if source:
631
+ os.environ["HARNESS_BACKGROUND_NOISE"] = source
632
+ else:
633
+ os.environ.pop("HARNESS_BACKGROUND_NOISE", None)
634
+ code = place_the_call(
635
+ os.environ.get("HARNESS_VOICE_CASE", "2.1.2"),
636
+ on_exchange=live_exchange if on_exchange else None,
637
+ )
638
+ finally:
639
+ try:
640
+ webhook.stop()
641
+ except Exception:
642
+ logger.exception(
643
+ "webhook cleanup failed after scenario %s", scenario.name
644
+ )
645
+ if tunnel is not None:
646
+ try:
647
+ tunnel.terminate()
648
+ except Exception:
649
+ logger.exception(
650
+ "tunnel cleanup failed after scenario %s", scenario.name
651
+ )
652
+ if (Path(world_root) / "environment.json").exists():
653
+ try:
654
+ from ..provision import runtime_logs, stop_runtime
655
+
656
+ # Some third-party LiveKit agents execute every tool in-process. They do not
657
+ # call the harness webhook and may not implement HARNESS_AGENT_TOOL_TRACE,
658
+ # but the LiveKit worker emits structured execution lifecycle events. Read
659
+ # those events before removing the per-scenario container. Raw logs are not
660
+ # retained; only normalized tool evidence is kept below.
661
+ runtime_output = runtime_logs(world_root)
662
+ stop_runtime(world_root)
663
+ except Exception:
664
+ logger.exception(
665
+ "runtime cleanup failed after scenario %s", scenario.name
666
+ )
667
+ # Everything the runner recorded about this call, read from the report it wrote.
668
+ return code, newest_report(started, root=sdk_output), runtime_output
669
+
670
+ attempts = 1 + max(0, int(os.environ.get("HARNESS_VOICE_INFRA_RETRIES", "1")))
671
+ code, case, runtime_output = 1, {}, ""
672
+ attempts_used = 0
673
+ trace_path = folder / "agent-tool-calls.jsonl"
674
+ for attempt in range(attempts):
675
+ attempts_used = attempt + 1
676
+ # Every attempt owns its trace. A stale line from a failed attempt must not turn a later
677
+ # worker-join failure into something that looks like agent evidence.
678
+ trace_path.unlink(missing_ok=True)
679
+ # The webhook is the transport-level evidence fallback. Clear calls from a failed voice
680
+ # attempt before retrying so only the attempt whose transcript is graded can contribute.
681
+ world.calls = []
682
+ code, case, runtime_output = await asyncio.to_thread(placed_once)
683
+ attempt_calls = _semantic_calls(
684
+ trace_path, contract=contract
685
+ ) or _livekit_log_calls(runtime_output, contract=contract)
686
+ if (
687
+ not _voice_attempt_should_retry(
688
+ code, case, has_agent_calls=bool(attempt_calls or world.calls)
689
+ )
690
+ or attempt + 1 >= attempts
691
+ ):
692
+ break
693
+ logger.warning(
694
+ "retryable voice attempt ended early for %s; retrying attempt %s/%s",
695
+ scenario.name,
696
+ attempt + 2,
697
+ attempts,
698
+ )
699
+ semantic = _semantic_calls(trace_path, contract=contract) or _livekit_log_calls(
700
+ runtime_output, contract=contract
701
+ )
702
+ # A worker trace includes semantic/local actions that never cross HTTP and is preferred when
703
+ # available. In a hosted Docker runner, however, the job artifacts can live in a named volume
704
+ # whose container path cannot be bind-mounted by the host daemon into the submitted runtime.
705
+ # In that case the bound world is still exact evidence: setup calls were cleared in prepare,
706
+ # caller hydration uses record=False, and every remaining call arrived through this call's
707
+ # private webhook. Do not erase that evidence merely because the optional trace is absent.
708
+ if semantic:
709
+ world.calls = semantic
710
+ spoken = str(case.get("transcript") or "")
711
+ # Every track that exists, copied in beside the result so a run is self-contained and the
712
+ # page can fall back when the preferred one is missing.
713
+ kept = _keep_tracks(tracks_in(case), folder)
714
+
715
+ if _voice_infrastructure_failure(code, case, has_agent_calls=bool(semantic)):
716
+ status = str((case.get("metadata") or {}).get("status") or "failed")
717
+ result = Result(
718
+ scenario=scenario.name,
719
+ tests=scenario.tests,
720
+ problems=[
721
+ "voice infrastructure failed after retry: the target worker never joined or "
722
+ "produced an assistant turn; this is not graded as an agent failure"
723
+ ],
724
+ ended=status,
725
+ turns=len([line for line in spoken.splitlines() if line.strip()]),
726
+ calls=0,
727
+ transcript=spoken,
728
+ recording=(kept[0]["path"] if kept else ""),
729
+ )
730
+ result.tracks = kept
731
+ result.measured = measured(case)
732
+ result.measured["voice_attempts"] = attempts_used
733
+ return result
734
+
735
+ catalogue = load_catalogue(world_root)
736
+ settled = grade_sub_goals(world, scenario, catalogue, world.calls)
737
+ # Judged the same way a typed run is. Without this a spoken scenario reports "1/2" when what
738
+ # happened is that one check passed and the other was never asked, which reads as the agent
739
+ # half-failing rather than as the suite not having looked.
740
+ spoken_transcript = Transcript(
741
+ exchanges=[_said(line) for line in spoken.splitlines() if line.strip()],
742
+ calls=list(world.calls),
743
+ ended=str((case.get("metadata") or {}).get("status") or "finished"),
744
+ )
745
+ judgements, judged_cost = await judge(
746
+ scenario,
747
+ spoken_transcript,
748
+ contract,
749
+ catalogue,
750
+ model=roles["judge"],
751
+ ending=", ".join(
752
+ f"{name}: {len(rows)} rows"
753
+ for name, rows in sorted(world.observe().state.items())
754
+ ),
755
+ )
756
+ _require_action_evidence(judgements, scenario, world.calls)
757
+ suite_judgements = judge_suite_evals(
758
+ catalogue.suite_evals,
759
+ scenario,
760
+ spoken_transcript,
761
+ contract,
762
+ ending=", ".join(
763
+ f"{name}: {len(rows)} rows"
764
+ for name, rows in sorted(world.observe().state.items())
765
+ ),
766
+ )
767
+ judgements += reconcile_task_completion(suite_judgements, settled, judgements)
768
+ result = Result(
769
+ scenario=scenario.name,
770
+ tests=scenario.tests,
771
+ problems=[
772
+ f"{name} is not in this catalogue, so nothing graded it"
773
+ for name in ungraded_sub_goals(scenario, catalogue)
774
+ ],
775
+ state_failures=[f"{one.name}: {one.said}" for one in settled if not one.held],
776
+ conduct=judgements,
777
+ checkpoints=checkpoints(settled, judgements),
778
+ spent_usd=judged_cost,
779
+ ended=str((case.get("metadata") or {}).get("status") or "finished"),
780
+ turns=len([line for line in spoken.splitlines() if line.strip()]),
781
+ calls=len(world.calls),
782
+ transcript=spoken,
783
+ exchanges=_timed_exchanges(spoken_transcript.exchanges, spoken_times(case)),
784
+ recording=(kept[0]["path"] if kept else ""),
785
+ )
786
+ result.tracks = kept
787
+ result.measured = measured(case)
788
+ result.measured["voice_attempts"] = attempts_used
789
+ result.calls_detail = _calls_of(world.calls)
790
+ return result
791
+
792
+
793
+ def _normalize_live_exchange(turn: dict[str, Any]) -> dict[str, Any]:
794
+ """Translate ALK's report roles to the harness UI's conversation roles.
795
+
796
+ The callback comes from the simulator's own AgentSession, so its roles are from that
797
+ session's point of view: ``assistant`` is the simulated customer and ``user`` is the tested
798
+ service agent. The completed report later translates them to the test's point of view.
799
+ """
800
+ raw = str(turn.get("speaker") or turn.get("role") or "").strip().lower()
801
+ speaker = (
802
+ "customer"
803
+ if raw in {"assistant", "customer"}
804
+ else "agent"
805
+ if raw in {"user", "agent"}
806
+ else raw or "customer"
807
+ )
808
+ return {
809
+ **turn,
810
+ "speaker": speaker,
811
+ "text": turn.get("text") or turn.get("content") or "",
812
+ }
813
+
814
+
815
+ def _require_action_evidence(
816
+ judgements: list[Judgement], scenario: Scenario, calls: list[Call]
817
+ ) -> None:
818
+ """Prevent conditional prose checks from passing vacuously when the agent did nothing.
819
+
820
+ A judge can reasonably say "surge was disclosed before confirmation" when neither event
821
+ occurred because the implication is technically vacuous. In an executable scenario with a
822
+ reference action sequence, no semantic calls means the scenario conduct was not satisfied.
823
+ """
824
+ if not scenario.solution or calls:
825
+ return
826
+ for judgement in judgements:
827
+ if judgement.holds:
828
+ judgement.holds = False
829
+ judgement.why = (
830
+ "The agent made no semantic tool calls, so this scenario conduct cannot be "
831
+ "credited even if its ordering condition is vacuously true."
832
+ )
833
+
834
+
835
+ def _voice_infrastructure_failure(
836
+ code: int, case: dict[str, Any], *, has_agent_calls: bool = False
837
+ ) -> bool:
838
+ """A failed room before meaningful target activity is not agent evidence."""
839
+ if code == 0:
840
+ return False
841
+ transcript = str(case.get("transcript") or "")
842
+ lines = [line for line in transcript.splitlines() if line.strip()]
843
+ has_target_turn = any(
844
+ line.strip().lower().startswith(("assistant:", "agent:")) for line in lines
845
+ )
846
+ if not has_target_turn:
847
+ return True
848
+ failure = (case.get("metadata") or {}).get("failure") or case.get("failure") or {}
849
+ retryable = bool(failure.get("retryable")) if isinstance(failure, dict) else False
850
+ failure_code = str(failure.get("code") or "") if isinstance(failure, dict) else ""
851
+ transport_retryable = retryable and failure_code in {
852
+ "target_disconnected",
853
+ "target_not_found",
854
+ "provider_disconnected",
855
+ "room_connection_failed",
856
+ "room_not_ready",
857
+ }
858
+ target_lines = [
859
+ line.split(":", 1)[-1].strip()
860
+ for line in lines
861
+ if line.strip().lower().startswith(("assistant:", "agent:"))
862
+ ]
863
+ truncated_target = any(
864
+ utterance and utterance[-1] not in ".?!" for utterance in target_lines
865
+ )
866
+ # ALK requires six alternating messages by default. A retryable disconnect before that,
867
+ # without one semantic action, is a room/worker lifecycle failure; a longer conversation or
868
+ # any tool trace is enough evidence to grade the agent normally.
869
+ return (
870
+ not has_agent_calls
871
+ and len(lines) < 6
872
+ and (transport_retryable or truncated_target)
873
+ )
874
+
875
+
876
+ def _voice_attempt_should_retry(
877
+ code: int, case: dict[str, Any], *, has_agent_calls: bool = False
878
+ ) -> bool:
879
+ """Retry one short silence without misclassifying the final result as infrastructure.
880
+
881
+ Live speech recognition occasionally drops the caller's first audio turn. The provider
882
+ labels that timeout retryable, but a greeting proves the worker joined, so if the retry also
883
+ fails it remains an agent-pipeline reliability failure. A bounded second attempt separates
884
+ a transient dropped turn from a reproducible weak branch while preserving both outcomes via
885
+ ``voice_attempts``.
886
+ """
887
+ if _voice_infrastructure_failure(code, case, has_agent_calls=has_agent_calls):
888
+ return True
889
+ if code == 0 or has_agent_calls:
890
+ return False
891
+ failure = (case.get("metadata") or {}).get("failure") or case.get("failure") or {}
892
+ if not isinstance(failure, dict):
893
+ return False
894
+ lines = [
895
+ line for line in str(case.get("transcript") or "").splitlines() if line.strip()
896
+ ]
897
+ return (
898
+ bool(failure.get("retryable"))
899
+ and str(failure.get("code") or "") == "conversation_silence_timeout"
900
+ and len(lines) < 6
901
+ )
902
+
903
+
904
+ def _semantic_calls(path: Path, *, contract: AgentContract | None = None) -> list[Call]:
905
+ """Read the submitted worker's agent-facing tool trace, tolerating a killed final line."""
906
+ if not path.exists():
907
+ return []
908
+ endpoint_names = {
909
+ entry.endpoint.strip("/"): entry.tool
910
+ for entry in (contract.tool_entrypoints if contract is not None else [])
911
+ if entry.endpoint.strip("/")
912
+ }
913
+ calls: list[Any] = []
914
+ for line in path.read_text(encoding="utf-8").splitlines():
915
+ try:
916
+ record = json.loads(line)
917
+ except (json.JSONDecodeError, TypeError):
918
+ continue
919
+ if not isinstance(record, dict) or not record.get("name"):
920
+ continue
921
+ output = record.get("output")
922
+ if isinstance(output, str):
923
+ try:
924
+ output = json.loads(output)
925
+ except json.JSONDecodeError:
926
+ pass
927
+ arguments = record.get("arguments") or {}
928
+ if isinstance(arguments, str):
929
+ try:
930
+ arguments = json.loads(arguments)
931
+ except json.JSONDecodeError:
932
+ arguments = {"raw": arguments}
933
+ if not isinstance(arguments, dict):
934
+ arguments = {"value": arguments}
935
+ failed = bool(record.get("is_error"))
936
+ recorded_name = str(record["name"]).strip("/")
937
+ call = Call(
938
+ name=endpoint_names.get(recorded_name, recorded_name),
939
+ arguments=arguments,
940
+ result=output,
941
+ ok=not failed,
942
+ refused=failed,
943
+ error=str(output) if failed else "",
944
+ at=float(record.get("at") or 0.0),
945
+ )
946
+ # A harness-aware worker may mirror a local state-machine action to the world for
947
+ # observability and then emit the authoritative function-completion event. They are one
948
+ # logical action. Prefer the completion result, but never collapse ordinary identical
949
+ # retries (payment-status polling is a legitimate example).
950
+ if calls and _telemetry_mirror(calls[-1], call):
951
+ calls[-1] = call
952
+ else:
953
+ calls.append(call)
954
+ return calls
955
+
956
+
957
+ _ANSI = re.compile(r"\x1b\[[0-?]*[ -/]*[@-~]")
958
+
959
+
960
+ def _livekit_log_calls(
961
+ output: str, *, contract: AgentContract | None = None
962
+ ) -> list[Call]:
963
+ """Normalize completed LiveKit Python tool executions from bounded runtime logs.
964
+
965
+ This fallback is intentionally narrow: a start event alone earns no evidence, and arbitrary
966
+ application prose is never interpreted as a call. LiveKit's structured ``executing tool``
967
+ and matching ``tools execution completed`` records are stable SDK lifecycle events. The
968
+ Successful result values are not present in those logs, so they are represented honestly as
969
+ completion evidence rather than fabricated output. Structured ``ToolError while executing
970
+ tool`` records do carry a refusal reason; correlate those with the matching start so a
971
+ truthful refusal is never normalized as success.
972
+ """
973
+ endpoint_names = {
974
+ entry.endpoint.strip("/"): entry.tool
975
+ for entry in (contract.tool_entrypoints if contract is not None else [])
976
+ if entry.endpoint.strip("/")
977
+ }
978
+ contract_tools = (
979
+ {tool.name: tool for tool in contract.tools} if contract is not None else {}
980
+ )
981
+ decoder = json.JSONDecoder()
982
+ starts: list[dict[str, Any]] = []
983
+ completed: set[str] = set()
984
+ refusals: dict[tuple[str, str], str] = {}
985
+ for raw_line in str(output or "").splitlines():
986
+ line = _ANSI.sub("", raw_line)
987
+ marker = "executing tool"
988
+ completion = "tools execution completed"
989
+ record: Any = None
990
+ try:
991
+ structured = json.loads(line)
992
+ except (json.JSONDecodeError, TypeError):
993
+ structured = None
994
+ structured_message = (
995
+ str(structured.get("message") or "") if isinstance(structured, dict) else ""
996
+ )
997
+ refusal = structured_message.startswith("ToolError while executing tool:")
998
+ if isinstance(structured, dict) and (
999
+ structured_message in {marker, completion} or refusal
1000
+ ):
1001
+ record = structured
1002
+ event = "refusal" if refusal else structured_message
1003
+ elif marker in line:
1004
+ brace = line.find("{", line.find(marker) + len(marker))
1005
+ if brace < 0:
1006
+ continue
1007
+ try:
1008
+ record, _ = decoder.raw_decode(line[brace:])
1009
+ except (json.JSONDecodeError, TypeError):
1010
+ continue
1011
+ event = marker
1012
+ elif completion in line:
1013
+ brace = line.find("{", line.find(completion) + len(completion))
1014
+ if brace < 0:
1015
+ continue
1016
+ try:
1017
+ record, _ = decoder.raw_decode(line[brace:])
1018
+ except (json.JSONDecodeError, TypeError):
1019
+ continue
1020
+ event = completion
1021
+ else:
1022
+ continue
1023
+ if not isinstance(record, dict):
1024
+ continue
1025
+ if event == "refusal":
1026
+ speech_id = str(record.get("speech_id") or "")
1027
+ function = str(record.get("function") or "").strip("/")
1028
+ if speech_id and function:
1029
+ refusals[(speech_id, function)] = structured_message.split(":", 1)[
1030
+ -1
1031
+ ].strip()
1032
+ elif event == marker:
1033
+ if not record.get("function"):
1034
+ continue
1035
+ speech_id = str(record.get("speech_id") or "")
1036
+ recorded_name = str(record["function"]).strip("/")
1037
+ name = endpoint_names.get(recorded_name, recorded_name)
1038
+ arguments: Any = record.get("lk.pii.arguments") or {}
1039
+ if isinstance(arguments, str):
1040
+ try:
1041
+ arguments = json.loads(arguments)
1042
+ except json.JSONDecodeError:
1043
+ arguments = {"raw": arguments}
1044
+ if not isinstance(arguments, dict):
1045
+ arguments = {"value": arguments}
1046
+ if contract is not None:
1047
+ specification = contract_tools.get(name)
1048
+ # LiveKit also logs SDK/workflow functions which are not target tools. Only the
1049
+ # contract can make a runtime-log fallback authoritative target evidence.
1050
+ if specification is None:
1051
+ continue
1052
+ # A speech-level completion has no per-tool result. Do not turn a malformed start
1053
+ # into success merely because the surrounding speech completed.
1054
+ if any(argument not in arguments for argument in specification.args):
1055
+ continue
1056
+ starts.append(
1057
+ {
1058
+ "name": name,
1059
+ "recorded_name": recorded_name,
1060
+ "arguments": arguments,
1061
+ "speech_id": speech_id,
1062
+ "at": _log_timestamp(str(record.get("timestamp") or line)),
1063
+ }
1064
+ )
1065
+ elif event == completion:
1066
+ if record.get("speech_id"):
1067
+ completed.add(str(record["speech_id"]))
1068
+ calls: list[Call] = []
1069
+ starts_per_speech = {
1070
+ speech_id: sum(one["speech_id"] == speech_id for one in starts)
1071
+ for speech_id in completed
1072
+ }
1073
+ for one in starts:
1074
+ if not one["speech_id"] or one["speech_id"] not in completed:
1075
+ continue
1076
+ # A single LiveKit completion can close a batch of tool starts but cannot prove which
1077
+ # individual invocation succeeded. Native semantic traces remain authoritative for that
1078
+ # case; the bounded-log fallback deliberately emits no ambiguous evidence.
1079
+ if starts_per_speech.get(one["speech_id"]) != 1:
1080
+ continue
1081
+ error = refusals.get(
1082
+ (one["speech_id"], one["recorded_name"]), ""
1083
+ ) or refusals.get((one["speech_id"], one["name"]), "")
1084
+ calls.append(
1085
+ Call(
1086
+ name=one["name"],
1087
+ arguments=one["arguments"],
1088
+ result=(
1089
+ error
1090
+ if error
1091
+ else {
1092
+ "evidence": "livekit_runtime_log",
1093
+ "execution": "completed",
1094
+ }
1095
+ ),
1096
+ ok=not error,
1097
+ refused=bool(error),
1098
+ error=error,
1099
+ at=float(one["at"]),
1100
+ )
1101
+ )
1102
+ return calls
1103
+
1104
+
1105
+ def _log_timestamp(line: str) -> float:
1106
+ value = line.strip()
1107
+ matched = re.match(r"^(\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2},\d{3})", value)
1108
+ if matched is not None:
1109
+ value = matched.group(1).replace(",", ".")
1110
+ try:
1111
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
1112
+ return (parsed if parsed.tzinfo else parsed.replace(tzinfo=UTC)).timestamp()
1113
+ except ValueError:
1114
+ return 0.0
1115
+
1116
+
1117
+ def _telemetry_mirror(previous: Call, current: Call) -> bool:
1118
+ if previous.name != current.name or previous.arguments != current.arguments:
1119
+ return False
1120
+ result = previous.result
1121
+ if isinstance(result, dict):
1122
+ if result.get("execution") == "submitted_agent_runtime":
1123
+ return True
1124
+ text = str(result.get("result") or "").lower()
1125
+ else:
1126
+ text = str(result or "").lower()
1127
+ return "submitted service has no endpoint" in text
1128
+
1129
+
1130
+ def _timed_exchanges(
1131
+ exchanges: list[Any], times: list[dict[str, Any]]
1132
+ ) -> list[dict[str, Any]]:
1133
+ """The conversation with each turn's speech times attached, where they were measured.
1134
+
1135
+ Paired by position, and only when the two agree on how many turns there were. They come
1136
+ from the same call but by different routes, so a mismatch means one of them dropped a turn
1137
+ -- and pairing them anyway would hang every turn's timing on the wrong words.
1138
+ """
1139
+ spoken = [{"speaker": turn.speaker, "text": turn.text} for turn in exchanges]
1140
+ if len(times) != len(spoken):
1141
+ return spoken
1142
+ for turn, when in zip(spoken, times, strict=True):
1143
+ if when.get("start_time_ms") is None:
1144
+ continue
1145
+ turn["start_time_ms"] = when["start_time_ms"]
1146
+ if when.get("end_time_ms") is not None:
1147
+ turn["end_time_ms"] = when["end_time_ms"]
1148
+ return spoken
1149
+
1150
+
1151
+ def _keep_tracks(found: list[dict[str, str]], folder: Path) -> list[dict[str, str]]:
1152
+ """Copy each recording into this run's folder, keeping the order it was offered in.
1153
+
1154
+ Copied rather than referenced, because the runner's own directory is transient and a run
1155
+ that cannot be listened to next week is a run that cannot be shown to anybody.
1156
+ """
1157
+ import shutil
1158
+
1159
+ folder.mkdir(parents=True, exist_ok=True)
1160
+ kept: list[dict[str, str]] = []
1161
+ for track in found:
1162
+ source = Path(track["path"])
1163
+ if not source.exists():
1164
+ continue
1165
+ landed = folder / f"{track['label'].replace(':', '_')}{source.suffix}"
1166
+ try:
1167
+ shutil.copyfile(source, landed)
1168
+ except OSError:
1169
+ continue
1170
+ kept.append({"label": track["label"], "path": str(landed)})
1171
+ return kept
1172
+
1173
+
1174
+ def _averaged(measured: list[dict[str, Any]]) -> list[dict[str, Any]]:
1175
+ """Each metric's mean over the scenarios that reported it, carrying whether it applied.
1176
+
1177
+ A metric that had nothing to measure scores 1.0, so averaging the lot produces a suite
1178
+ summary in which two thirds of the numbers are perfect and none of them mean anything. The
1179
+ applicability travels with the average instead of being flattened away, so a reader is never
1180
+ shown "browser action safety 1.00" for a suite of phone calls without also being told there
1181
+ were no browser actions.
1182
+ """
1183
+ gathered: dict[str, list[float]] = {}
1184
+ applies: dict[str, bool] = {}
1185
+ reasons: dict[str, str] = {}
1186
+ for one in measured:
1187
+ for metric in (one or {}).get("metrics") or []:
1188
+ name, value = metric.get("name"), metric.get("score")
1189
+ if not name or not isinstance(value, (int, float)):
1190
+ continue
1191
+ gathered.setdefault(name, []).append(float(value))
1192
+ # Applicable anywhere is applicable: one scenario exercising a capability is enough
1193
+ # to make the number worth reading across the suite.
1194
+ applies[name] = applies.get(name, False) or bool(
1195
+ metric.get("applicable", True)
1196
+ )
1197
+ if metric.get("reason") and name not in reasons:
1198
+ reasons[name] = str(metric["reason"])
1199
+ return [
1200
+ {
1201
+ "name": name,
1202
+ "score": round(sum(values) / len(values), 4),
1203
+ "applicable": applies.get(name, True),
1204
+ "reason": reasons.get(name, ""),
1205
+ "cases": len(values),
1206
+ }
1207
+ for name, values in sorted(gathered.items())
1208
+ if values
1209
+ ]