agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,601 @@
1
+ """The tools that run a scenario against the real agent, and record what happened.
2
+
3
+ Placing a call was a command before this existed, which made the last stage the only one you
4
+ could not simply ask for. Nothing about it needed to be a command: wiring the world to the
5
+ assistant and grading afterwards is already code, and choosing which scenario to run and reading
6
+ what came back is the part worth having judgement on.
7
+
8
+ So the same shape as every other stage. The tools do what must be exact — restore the world,
9
+ repoint the assistant's own tools, place the call through ALK, run the checks — and the stage
10
+ decides what to run and says what it means.
11
+
12
+ A run takes minutes, not seconds. The tool blocks for that long, and says so, because a stage
13
+ that fires a call and returns immediately would report on a conversation that has not happened.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import asyncio
19
+ import json
20
+ import os
21
+ import re
22
+ import shutil
23
+ import time
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ from ..backends import tool, tool_server
28
+
29
+ from .. import platform
30
+ from ..catalogue import load_catalogue
31
+ from ..config import ARTIFACTS_ROOT
32
+ from ..scenario_tools import load_scenarios
33
+ from ..tools import schema
34
+ from ..world.snapshot import require_source_implementation
35
+ from .call import CASE, place_the_call
36
+ from .live import LiveRun, grade, wire
37
+
38
+ RUN_SERVER = "runs"
39
+ RESULTS = "runs.json"
40
+
41
+ # What a call needs before it can be placed at all, per transport. Checked up front rather than
42
+ # three minutes in, because the failure otherwise arrives after the expensive part.
43
+ #
44
+ # Which transport is in play is decided by the case: the 1.x cases reach a LiveKit worker, the
45
+ # 2.x cases a hosted Vapi assistant. Asking for the other one's credentials is how a working
46
+ # setup gets reported as broken.
47
+ REQUIRED_VAPI = ("VAPI_API_KEY", "VAPI_ASSISTANT_ID")
48
+ REQUIRED_LIVEKIT = (
49
+ "LIVEKIT_API_KEY",
50
+ "LIVEKIT_API_SECRET",
51
+ "LIVEKIT_TARGET_AGENT_NAME",
52
+ )
53
+
54
+
55
+ def _livekit_case() -> bool:
56
+ """Whether the case being run reaches a LiveKit worker rather than a hosted assistant."""
57
+ return os.environ.get("HARNESS_VOICE_CASE", CASE).strip().startswith("1.")
58
+
59
+
60
+ def _ok(text: str) -> dict[str, Any]:
61
+ return {"content": [{"type": "text", "text": text}]}
62
+
63
+
64
+ def _err(text: str) -> dict[str, Any]:
65
+ return {"content": [{"type": "text", "text": text}], "is_error": True}
66
+
67
+
68
+ def configure_source_voice(world_root: Path, contract: Any = None) -> bool:
69
+ """Select and configure the zero-setup voice path for a provisioned source runtime."""
70
+ root = Path(world_root)
71
+ if not (root / "environment.json").exists():
72
+ return False
73
+ from ..provision import activate_voice_environment
74
+
75
+ prompt = str(getattr(contract, "system_prompt_excerpt", "") or "").strip()
76
+ activate_voice_environment(root, system_prompt=prompt)
77
+ if not os.environ.get("LIVEKIT_TARGET_AGENT_NAME", "").strip():
78
+ label = str(getattr(contract, "agent", "") or "harness-agent").lower()
79
+ label = re.sub(r"[^a-z0-9-]+", "-", label).strip("-") or "harness-agent"
80
+ # The worker setup replaces this with a scenario-scoped value immediately before start.
81
+ # This base merely lets valid unnamed AgentServer workers pass transport preflight.
82
+ os.environ["LIVEKIT_TARGET_AGENT_NAME"] = label[:48]
83
+ os.environ.setdefault("HARNESS_VOICE_CASE", "1.1.2")
84
+ # This is an inbound worker: it greets first. The scripted caller turns that one greeting
85
+ # into its opening request. Simulator-first races an independent opening timer against the
86
+ # greeting transcript and can emit two consecutive caller turns.
87
+ os.environ["HARNESS_CONVERSATION_DIRECTION"] = "agent_first"
88
+ return True
89
+
90
+
91
+ def missing_prerequisites(
92
+ world_root: Path | None = None, contract: Any = None
93
+ ) -> list[str]:
94
+ """What would stop a live call, in the words of what to do about it."""
95
+ source_backed = bool(world_root and configure_source_voice(world_root, contract))
96
+ problems: list[str] = []
97
+ livekit = _livekit_case()
98
+ absent = [
99
+ name
100
+ for name in (REQUIRED_LIVEKIT if livekit else REQUIRED_VAPI)
101
+ if not os.environ.get(name)
102
+ ]
103
+ if absent:
104
+ suffix = (
105
+ "The submitted runtime does not provide them and the platform has no workspace "
106
+ "values configured."
107
+ if source_backed
108
+ else "Load the environment that owns these credentials before running."
109
+ )
110
+ problems.append(
111
+ f"{', '.join(absent)} not set, so there is no way to reach the agent. {suffix}"
112
+ )
113
+ credential = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS", "").strip()
114
+ google_simulator = os.environ.get("SIMULATOR_LLM_PROVIDER", "google").lower() in {
115
+ "google",
116
+ "gemini",
117
+ "vertex",
118
+ }
119
+ if livekit and google_simulator and credential and not Path(credential).is_file():
120
+ problems.append(
121
+ "GOOGLE_APPLICATION_CREDENTIALS points to a file that does not exist in the "
122
+ "harness runtime. Configure the platform's simulator credential mount once; this "
123
+ "is not per-agent setup."
124
+ )
125
+ # A LiveKit worker we run ourselves calls the world directly on the network we share with it,
126
+ # so there is nothing to expose. Only a hosted assistant has to reach in from outside.
127
+ exposed = os.environ.get("HARNESS_WEBHOOK_URL") or shutil.which("cloudflared")
128
+ if not livekit and not exposed:
129
+ problems.append(
130
+ "no way to expose the webhook publicly. A hosted agent cannot reach loopback, so "
131
+ "either install cloudflared (brew install cloudflared) or set HARNESS_WEBHOOK_URL "
132
+ "to a tunnel that is already running."
133
+ )
134
+ return problems
135
+
136
+
137
+ def save_results(results: list[dict[str, Any]], destination: Path) -> Path:
138
+ """Keep every run, so a suite can be read after the fact rather than scrolled back to."""
139
+ destination = Path(destination)
140
+ destination.mkdir(parents=True, exist_ok=True)
141
+ path = destination / RESULTS
142
+ path.write_text(json.dumps(results, indent=2, ensure_ascii=False), encoding="utf-8")
143
+ return path
144
+
145
+
146
+ def load_results(destination: Path) -> list[dict[str, Any]]:
147
+ path = Path(destination) / RESULTS
148
+ if not path.exists():
149
+ return []
150
+ try:
151
+ loaded = json.loads(path.read_text(encoding="utf-8"))
152
+ return loaded if isinstance(loaded, list) else []
153
+ except json.JSONDecodeError:
154
+ return []
155
+
156
+
157
+ def as_record(run: LiveRun) -> dict[str, Any]:
158
+ return {
159
+ "scenario": run.scenario,
160
+ "passed": bool(run.settled)
161
+ and run.met == len(run.settled)
162
+ and not run.problems,
163
+ "met": run.met,
164
+ "of": len(run.settled),
165
+ "settled": [
166
+ {"name": one.name, "held": one.held, "said": one.said, "broken": one.broken}
167
+ for one in run.settled
168
+ ],
169
+ "judged": list(run.judged),
170
+ "calls": list(run.calls),
171
+ "problems": list(run.problems),
172
+ }
173
+
174
+
175
+ def transcript_since(started: float) -> str:
176
+ """What was said on the call that just happened, from the voice runner's own report.
177
+
178
+ The voice case owns the call and writes its report where it always has; reaching into that
179
+ report is how the transcript gets onto the run record without the harness re-implementing
180
+ any of the call. Only a report written after this run started counts — the newest file on
181
+ disk is otherwise last week's call wearing today's verdict.
182
+ """
183
+ root = ARTIFACTS_ROOT / "simulation-acceptance"
184
+ if not root.exists():
185
+ return ""
186
+ newest: tuple[float, Path] | None = None
187
+ for report in root.glob("run_*/*/report.json"):
188
+ written = report.stat().st_mtime
189
+ if written >= started and (newest is None or written > newest[0]):
190
+ newest = (written, report)
191
+ if newest is None:
192
+ return ""
193
+ try:
194
+ loaded = json.loads(newest[1].read_text(encoding="utf-8"))
195
+ except (json.JSONDecodeError, OSError):
196
+ return ""
197
+ for result in loaded.get("results") or []:
198
+ spoken = result.get("transcript")
199
+ if isinstance(spoken, str) and spoken.strip():
200
+ return spoken
201
+ return ""
202
+
203
+
204
+ def report(run: LiveRun) -> str:
205
+ """One run, as something worth reading rather than a score."""
206
+ lines = [run.line()]
207
+ lines += [one.line() for one in run.settled]
208
+ lines += [f" [?] {name} — judged, not settled by code" for name in run.judged]
209
+ if run.problems:
210
+ lines += [f" !! {problem}" for problem in run.problems]
211
+ lines.append("")
212
+ lines.append("what the agent actually did:")
213
+ lines += [
214
+ f" {call}" for call in run.calls or ["(no tool calls reached the world)"]
215
+ ]
216
+ return "\n".join(lines)
217
+
218
+
219
+ def run_tools(
220
+ world_root: Path,
221
+ destination: Path,
222
+ *,
223
+ contract: Any = None,
224
+ case: str = "",
225
+ ) -> Any:
226
+ """A server for running one agent's scenarios against the real thing.
227
+
228
+ How a scenario runs is decided by what the agent is, not by this stage. A hosted voice agent
229
+ gets the live path — its own tools repointed at the world over a webhook, the call placed
230
+ through ALK. Anything else runs here: the agent stood up from its contract, conversing over
231
+ the same world, graded by the same checks. The scenarios, the world and the grading are
232
+ identical either way; only the transport changes.
233
+ """
234
+ written = load_scenarios(destination)
235
+ catalogue = load_catalogue(destination)
236
+ results = load_results(destination)
237
+ live = bool(contract is not None and getattr(contract, "modality", "") == "voice")
238
+ if live:
239
+ configure_source_voice(world_root, contract)
240
+ voice_case = case or os.environ.get("HARNESS_VOICE_CASE", "2.1.2")
241
+ authenticity_error = ""
242
+ try:
243
+ require_source_implementation(world_root)
244
+ except (FileNotFoundError, RuntimeError) as failed:
245
+ authenticity_error = str(failed)
246
+
247
+ @tool(
248
+ "list_scenarios",
249
+ "The scenarios that can be run, what each one tests, and which of its sub-goals are "
250
+ "settled by code rather than left to a judge.",
251
+ schema({}, []),
252
+ )
253
+ async def list_scenarios(_args: dict[str, Any]) -> dict[str, Any]:
254
+ if not written:
255
+ return _err("no scenarios have been written for this agent yet")
256
+ lines: list[str] = []
257
+ for one in written:
258
+ settled = [
259
+ name
260
+ for name in one.sub_goals
261
+ if (found := catalogue.named(name)) and found.deterministic()
262
+ ]
263
+ judged = [name for name in one.sub_goals if name not in settled]
264
+ ran = next((r for r in results if r["scenario"] == one.name), None)
265
+ mark = (
266
+ ""
267
+ if ran is None
268
+ else (
269
+ " [last run: PASS]" if ran.get("passed") else " [last run: FAIL]"
270
+ )
271
+ )
272
+ lines.append(
273
+ f"{one.name}{mark}\n passes when: {one.tests or one.use_case or '—'}\n"
274
+ f" settled by code: {', '.join(settled) or 'none'}\n"
275
+ f" judged: {', '.join(judged) or 'none'}"
276
+ )
277
+ return _ok("\n".join(lines))
278
+
279
+ @tool(
280
+ "preflight",
281
+ "Check everything a run needs before spending one. For a hosted voice agent that is the "
282
+ "assistant's credentials and a way to expose the webhook publicly; for anything else "
283
+ "the run happens here and needs nothing external. Run this before the first run.",
284
+ schema({}, []),
285
+ )
286
+ async def preflight(_args: dict[str, Any]) -> dict[str, Any]:
287
+ if authenticity_error:
288
+ return _err(f"Not ready:\n - {authenticity_error}")
289
+ if not live:
290
+ return _err(
291
+ "Not ready: no shipped-runtime target is registered for this agent. The old "
292
+ "local target reconstructed it from the contract and is intentionally disabled."
293
+ )
294
+ problems = missing_prerequisites(world_root, contract)
295
+ if problems:
296
+ return _err("Not ready:\n - " + "\n - ".join(problems))
297
+ return _ok(
298
+ "Ready. Credentials are set and the webhook can be exposed. "
299
+ f"{len(written)} scenarios are available."
300
+ )
301
+
302
+ async def _run_here(scenario: Any) -> dict[str, Any]:
303
+ """The scenario against the agent stood up from its contract, over the same world."""
304
+ from . import run_suite
305
+
306
+ if authenticity_error:
307
+ return _err(authenticity_error)
308
+ if contract is None:
309
+ return _err("no contract is loaded, so there is no agent to stand up")
310
+ graded = await run_suite([scenario], contract, world_root, out=destination)
311
+ results[:] = load_results(destination)
312
+ result = graded[0]
313
+ lines = [result.line()] + [check.line() for check in result.checkpoints]
314
+ if result.transcript:
315
+ lines += ["", "the conversation:", result.transcript]
316
+ answer = "\n".join(lines)
317
+ return _ok(answer) if result.passed else _err(answer)
318
+
319
+ @tool(
320
+ "run_simulation",
321
+ "Run the whole suite. One call: every scenario, each in its own copy of the world, "
322
+ "graded, and written out as one run you can come back to.\n\n"
323
+ "This is how a suite is run. Running scenarios one at a time is for looking into a "
324
+ "single failure afterwards, not for getting results.\n\n"
325
+ "`concurrency` is how many run at once. Leave it at 1 for a spoken agent, where every "
326
+ "scenario is a real call. It takes minutes and blocks until the whole suite is done.",
327
+ schema({"concurrency": int, "model": str}, []),
328
+ )
329
+ async def run_simulation(args: dict[str, Any]) -> dict[str, Any]:
330
+ from .simulation import simulate
331
+
332
+ if authenticity_error:
333
+ return _err(authenticity_error)
334
+ if not live:
335
+ return _err(
336
+ "no shipped-runtime target is registered for this agent; refusing to "
337
+ "reconstruct it from the contract"
338
+ )
339
+ if contract is None:
340
+ return _err("no contract is loaded, so there is no agent to run against")
341
+ if not written:
342
+ return _err("there are no scenarios to run")
343
+ # Kept as they finish, because reporting needs the graded results themselves and the
344
+ # summary carries only their rendering.
345
+ produced: list[Any] = []
346
+ summary = await simulate(
347
+ list(written),
348
+ contract,
349
+ world_root,
350
+ destination=destination,
351
+ model=str(args.get("model") or "") or None,
352
+ concurrency=max(1, int(args.get("concurrency") or 1)),
353
+ on_case_done=produced.append,
354
+ )
355
+ results[:] = load_results(destination)
356
+ lines = [
357
+ f"{summary['run_id']}: {summary['passed']}/{summary['scenarios']} passed "
358
+ f"in {summary['seconds']}s, ${summary['spent_usd']}",
359
+ "",
360
+ ]
361
+ for one in summary["results"]:
362
+ mark = "PASS" if one["passed"] else "FAIL"
363
+ note = f" {one['problems'][0]}" if one["problems"] else ""
364
+ audio = " [recording]" if one["recording"] else ""
365
+ lines.append(
366
+ f" {mark} {one['scenario']} {one['met']}/{one['of']}{audio}{note}"
367
+ )
368
+ # Reported here too, not only from the run button: a run that reaches the platform only
369
+ # when it was started one particular way leaves the page an unreliable record of what
370
+ # has been run.
371
+ _, said = platform.deliver(
372
+ produced, list(written), destination, modality=contract.modality or "text"
373
+ )
374
+ lines += ["", *said]
375
+ lines += [
376
+ "",
377
+ "read_run gives any one of these in full: the conversation, every tool call with "
378
+ "its arguments, and what each check decided.",
379
+ ]
380
+ return _ok("\n".join(lines))
381
+
382
+ @tool(
383
+ "read_run",
384
+ "One run in full, or the list of runs when no id is given. A run holds every scenario's "
385
+ "conversation, every tool call with its arguments and result, and what each check "
386
+ "decided — which is what a failure is diagnosed from.",
387
+ schema({"run_id": str, "scenario": str}, []),
388
+ )
389
+ async def read_run(args: dict[str, Any]) -> dict[str, Any]:
390
+ from .simulation import every_run, read_run as load_run
391
+
392
+ run_id = str(args.get("run_id") or "")
393
+ if not run_id:
394
+ runs = every_run(destination)
395
+ if not runs:
396
+ return _ok("No runs yet. run_simulation makes one.")
397
+ return _ok(
398
+ "\n".join(
399
+ f" {one['run_id']} {one.get('passed', 0)}/{one.get('scenarios', 0)} "
400
+ f"passed {one.get('seconds', 0)}s"
401
+ for one in runs
402
+ )
403
+ )
404
+ try:
405
+ whole = load_run(destination, run_id)
406
+ except FileNotFoundError as missing:
407
+ return _err(str(missing))
408
+ wanted = str(args.get("scenario") or "")
409
+ cases = [
410
+ one
411
+ for one in whole.get("scenarios", [])
412
+ if not wanted or one.get("scenario") == wanted
413
+ ]
414
+ if not cases:
415
+ return _err(f"{run_id} has no scenario called {wanted!r}")
416
+ return _ok(json.dumps(cases if wanted else whole, indent=2, default=str)[:6000])
417
+
418
+ @tool(
419
+ "run_scenario",
420
+ "Run one scenario against the agent and grade it.\n\n"
421
+ "The world is restored and the scenario's setup applied first. A hosted voice agent is "
422
+ "reached live — its OWN tools are pointed at the world over a webhook and the call is "
423
+ "placed; any other agent is stood up here from its contract and conversed with. Either "
424
+ "way the sub-goals' checks run against what the world holds afterwards plus the calls "
425
+ "that were made.\n\n"
426
+ "It can take minutes and blocks until the run is over. Run one at a time and read what "
427
+ "comes back before running the next.",
428
+ # Both spellings accepted: every model that has driven this stage has guessed
429
+ # `scenario` at least once, and a retry on an argument name is a wasted turn.
430
+ schema({"name": str, "scenario": str}, []),
431
+ )
432
+ async def run_scenario(args: dict[str, Any]) -> dict[str, Any]:
433
+ if authenticity_error:
434
+ return _err(authenticity_error)
435
+ name = str(args.get("name") or args.get("scenario") or "")
436
+ scenario = next((one for one in written if one.name == name), None)
437
+ if scenario is None:
438
+ return _err(
439
+ f"no scenario called {name!r}. There is: "
440
+ + ", ".join(one.name for one in written)
441
+ )
442
+ if not live:
443
+ return await _run_here(scenario)
444
+ problems = missing_prerequisites(world_root, contract)
445
+ if problems:
446
+ return _err(
447
+ "Cannot place a call:\n - "
448
+ + "\n - ".join(problems)
449
+ + "\nThis is the environment this harness is running in, not something to fix "
450
+ "in the scenario."
451
+ )
452
+
453
+ def placed() -> tuple[LiveRun, str, list[str], str]:
454
+ """The whole call, off the event loop.
455
+
456
+ Wiring reads a subprocess's stdout and placing the call blocks for minutes; run
457
+ inline they freeze whatever loop is hosting this tool, which for the web UI means
458
+ the stream, the status endpoint and the stop button all die for the duration.
459
+ """
460
+ world, instruction, webhook, tunnel, url, moved = wire(scenario, world_root)
461
+ started = time.time()
462
+ try:
463
+ # The caller's instruction reaches the voice case through the environment, so
464
+ # how a simulated caller behaves is not decided in two places.
465
+ os.environ["HARNESS_INSTRUCTION"] = instruction
466
+ os.environ["HARNESS_SCENARIO"] = scenario.name
467
+ # The caller is never handed the grader's pass question. `tests` is written about
468
+ # the agent in the third person, so as an objective it reads as a rubric rather
469
+ # than a motive. What this person wants is already in the instruction.
470
+ os.environ.pop("HARNESS_OUTCOME", None)
471
+ os.environ["HARNESS_PERSONA"] = json.dumps(
472
+ scenario.persona.model_dump(exclude_none=True)
473
+ if scenario.persona is not None
474
+ else {"name": "customer"}
475
+ )
476
+ os.environ["HARNESS_INITIAL_MESSAGE"] = (
477
+ scenario.persona.initial_message
478
+ if scenario.persona is not None
479
+ else ""
480
+ )
481
+ code = place_the_call(voice_case)
482
+ run = grade(scenario, world, world_root)
483
+ if code != 0 and not run.calls:
484
+ run.problems.append(
485
+ f"the voice runner exited {code} and no tool call reached the world, "
486
+ "so this says nothing about the agent"
487
+ )
488
+ finally:
489
+ webhook.stop()
490
+ if tunnel is not None:
491
+ tunnel.terminate()
492
+ if (Path(world_root) / "environment.json").exists():
493
+ from ..provision import stop_runtime
494
+
495
+ stop_runtime(world_root)
496
+ world.close()
497
+ return run, url, moved, transcript_since(started)
498
+
499
+ run, url, moved, spoken = await asyncio.to_thread(placed)
500
+
501
+ record = as_record(run)
502
+ record["instruction"] = scenario.instruction
503
+ record["transcript"] = spoken
504
+ # Re-read before writing: the local suite writes the same file, and a list loaded when
505
+ # this stage opened would silently roll back anything recorded since.
506
+ results[:] = [
507
+ r for r in load_results(destination) if r.get("scenario") != scenario.name
508
+ ]
509
+ results.append(record)
510
+ save_results(results, destination)
511
+ answer = f"webhook: {url}/tool\nrepointed: {', '.join(moved)}\n\n{report(run)}"
512
+ return _ok(answer) if not run.problems else _err(answer)
513
+
514
+ @tool(
515
+ "read_results",
516
+ "What every scenario did the last time it was run, without running anything.",
517
+ schema({}, []),
518
+ )
519
+ async def read_results(_args: dict[str, Any]) -> dict[str, Any]:
520
+ if not results:
521
+ return _ok("nothing has been run yet")
522
+ lines = []
523
+ for record in results:
524
+ mark = "PASS" if record.get("passed") else "FAIL"
525
+ # Two record shapes share this file: live runs carry settled/judged, local runs
526
+ # carry checkpoints. Both say what failed, and both deserve to be read.
527
+ failed = [
528
+ f"{one.get('name')}: {one.get('said') or one.get('detail') or ''}"
529
+ for one in (record.get("settled") or record.get("checkpoints") or [])
530
+ if not (one.get("held") if "held" in one else one.get("passed"))
531
+ ]
532
+ met = record.get("met", record.get("checkpoints_met", "?"))
533
+ of = record.get("of")
534
+ scored = f"{met}/{of}" if of is not None else str(met)
535
+ lines.append(
536
+ f"{mark} {record.get('scenario')} {scored}"
537
+ + ("\n - " + "\n - ".join(failed) if failed else "")
538
+ )
539
+ passed = sum(1 for record in results if record.get("passed"))
540
+ return _ok("\n".join(lines) + f"\n\n{passed} of {len(results)} passed")
541
+
542
+ server = tool_server(
543
+ name=RUN_SERVER,
544
+ version="0.1.0",
545
+ tools=[
546
+ list_scenarios,
547
+ preflight,
548
+ run_simulation,
549
+ read_run,
550
+ run_scenario,
551
+ read_results,
552
+ ],
553
+ )
554
+ return server
555
+
556
+
557
+ TOOL_NAMES = (
558
+ "list_scenarios",
559
+ "preflight",
560
+ "run_simulation",
561
+ "read_run",
562
+ "run_scenario",
563
+ "read_results",
564
+ )
565
+
566
+
567
+ # Which of the several recordings a call leaves behind is the one worth keeping. Both sides on
568
+ # one track, because the question asked of a spoken run is nearly always about the interaction:
569
+ # whether the agent talked over the caller, how long it left them waiting, what it heard.
570
+ PREFERRED = ("_stereo.wav", "stereo.wav", "combined.wav")
571
+
572
+
573
+ def recording_since(started: float, into: Path) -> str:
574
+ """Copy the audio from the call that just happened into this run's folder.
575
+
576
+ ALK records already and writes several tracks under its own artifacts directory. Rather than
577
+ tell it where to put them — which it takes from its manifest, not from the environment — the
578
+ files it wrote are found the same way the transcript is, by being newer than the moment this
579
+ run began, and the one worth keeping is copied in beside the result.
580
+ """
581
+ root = ARTIFACTS_ROOT / "simulation-acceptance"
582
+ if not root.exists():
583
+ return ""
584
+ fresh = [
585
+ path
586
+ for path in root.rglob("*")
587
+ if path.is_file()
588
+ and path.suffix.lower() in (".wav", ".mp3", ".ogg")
589
+ and path.stat().st_mtime >= started
590
+ ]
591
+ if not fresh:
592
+ return ""
593
+ chosen = next(
594
+ (one for mark in PREFERRED for one in fresh if one.name.endswith(mark)),
595
+ max(fresh, key=lambda one: one.stat().st_size),
596
+ )
597
+ into = Path(into)
598
+ into.mkdir(parents=True, exist_ok=True)
599
+ landed = into / f"recording{chosen.suffix}"
600
+ shutil.copyfile(chosen, landed)
601
+ return str(landed)