agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,296 @@
1
+ """Stage four: run the scenarios against the world and say what happened.
2
+
3
+ Every scenario gets its own world. It is restored from the frozen snapshot, the scenario's own
4
+ setup is run against it, and it is thrown away afterwards. Nothing a scenario does can reach the
5
+ next one, which is what makes a result mean something on its own and makes the whole suite
6
+ repeatable a week later.
7
+
8
+ The shape is the same regardless of what is being tested: restore, converse, grade against the
9
+ state that is left behind. Where the agent actually runs is a target, so the same scenarios grade
10
+ a hosted agent without any of this changing.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ from pathlib import Path
17
+ from typing import Any, Callable, Sequence
18
+
19
+ from ..contract import AgentContract
20
+ from ..catalogue import load_catalogue
21
+ from ..simulator import load_simulator_prompt
22
+ from ..scenario import Scenario
23
+ from ..folder import apply_setup, check_ready
24
+ from ..world.snapshot import restore
25
+ from .conversation import FINISHED, Exchange, Transcript, converse
26
+ from .grade import (
27
+ Checkpoint,
28
+ Result,
29
+ as_json,
30
+ checkpoints,
31
+ grade_sub_goals,
32
+ judge,
33
+ judge_suite_evals,
34
+ summarise,
35
+ ungraded_sub_goals,
36
+ )
37
+ from .targets import (
38
+ LocalAgent,
39
+ RepositoryChatTarget,
40
+ Target,
41
+ register_target,
42
+ resolve,
43
+ supported,
44
+ )
45
+
46
+
47
+ def _cases(report: Any) -> list[Any]:
48
+ """The per-scenario results, whichever report this is.
49
+
50
+ The runner hands back a ``SimulationReport``, whose cases are ``test_cases`` and whose
51
+ messages live a level down; the plugins hand back the older ``TestReport``, whose cases are
52
+ ``results``. Reading only one of them finds nothing in the other and reports a run in which
53
+ nobody said anything, which is indistinguishable from an agent that ignored the person.
54
+ """
55
+ legacy = getattr(report, "to_legacy", None)
56
+ if callable(legacy):
57
+ try:
58
+ report = legacy()
59
+ except Exception: # noqa: BLE001 - a report that will not convert is still readable
60
+ pass
61
+ return list(
62
+ getattr(report, "results", None) or getattr(report, "test_cases", None) or []
63
+ )
64
+
65
+
66
+ def from_alk(report: Any, world, spent: float) -> Transcript:
67
+ """What ALK's report says happened, in the shape the grading already reads.
68
+
69
+ Read from ``messages``, the normalised trajectory, and not from ``transcript`` — which is a
70
+ string, so iterating it yields characters, matches nothing, and produces a run with no turns
71
+ at all. The judge is then handed an empty conversation and fails every claim about what was
72
+ said, which arrives looking like an agent that never spoke.
73
+ """
74
+ exchanges: list[Exchange] = []
75
+ for case in _cases(report):
76
+ for message in getattr(case, "messages", None) or []:
77
+ if not isinstance(message, dict):
78
+ continue
79
+ role = str(message.get("role") or "")
80
+ text = message.get("content") or ""
81
+ # Tool turns are in here too, and they are already recorded as calls. Putting them
82
+ # in the conversation as well would have the judge read a tool result as something
83
+ # the agent said.
84
+ if role in ("tool", "function") or not str(text).strip():
85
+ continue
86
+ exchanges.append(
87
+ Exchange("agent" if role == "assistant" else "customer", str(text))
88
+ )
89
+ if not exchanges and isinstance(getattr(case, "transcript", None), str):
90
+ # A plugin that only fills the text form still has to be readable.
91
+ spoken = case.transcript.strip()
92
+ if spoken:
93
+ exchanges.append(Exchange("customer", spoken))
94
+ return Transcript(
95
+ exchanges=exchanges,
96
+ calls=list(world.calls),
97
+ ended=FINISHED,
98
+ spent_usd=spent,
99
+ )
100
+
101
+
102
+ def audio_from(report: Any) -> str:
103
+ """Where ALK left this run's audio, if it left any.
104
+
105
+ Artifacts are how a modality hands back what it produced, so the recording is asked of the
106
+ report rather than guessed at from a directory. A run with none says so.
107
+ """
108
+ for case in _cases(report):
109
+ for artifact in getattr(case, "artifacts", None) or []:
110
+ kind = str(getattr(artifact, "type", "") or "")
111
+ mime = str(getattr(artifact, "mime_type", "") or "")
112
+ if "audio" in kind.lower() or mime.startswith("audio/"):
113
+ found = getattr(artifact, "path", None) or getattr(
114
+ artifact, "uri", None
115
+ )
116
+ if found:
117
+ return str(found)
118
+ return ""
119
+
120
+
121
+ RUNS = "runs.json"
122
+ REPORT = "report.txt"
123
+
124
+ __all__ = [
125
+ "Checkpoint",
126
+ "Exchange",
127
+ "LocalAgent",
128
+ "RepositoryChatTarget",
129
+ "Result",
130
+ "Target",
131
+ "Transcript",
132
+ "converse",
133
+ "register_target",
134
+ "run_scenario",
135
+ "run_suite",
136
+ "supported",
137
+ "summarise",
138
+ ]
139
+
140
+
141
+ async def run_scenario(
142
+ scenario: Scenario,
143
+ contract: AgentContract,
144
+ world_root: Path,
145
+ *,
146
+ target: str = "local",
147
+ model: str | None = None,
148
+ through_alk: bool = False,
149
+ on_exchange: Callable[[Exchange], Any] | None = None,
150
+ ) -> Result:
151
+ """Run one scenario in its own copy of the world and grade what it left behind."""
152
+ catalogue = load_catalogue(world_root)
153
+ world = restore(world_root)
154
+ try:
155
+ # reset() is how an environment is started in ALK: it clears the call log and
156
+ # publishes the tools and the starting state. Going through it keeps a generated world
157
+ # drivable by anything that already drives an environment.
158
+ world.reset()
159
+ applied = apply_setup(scenario, world)
160
+ if not applied.ok:
161
+ raise RuntimeError(f"the scenario's setup did not run: {applied.said}")
162
+ ready = check_ready(scenario, world)
163
+ if not ready.ok:
164
+ raise RuntimeError(
165
+ f"the world is not ready for this scenario: {ready.said}. Running it would "
166
+ "test us rather than the agent."
167
+ )
168
+ # The setup's calls are not the agent's.
169
+ world.calls = []
170
+ if through_alk:
171
+ # ALK owns the simulation and drives the world through EnvironmentAdapter; the
172
+ # harness only grades what it is left with. Nothing here is modality-specific,
173
+ # which is the point: the browser and voice runners take the same adapter.
174
+ from .alk import simulate
175
+
176
+ report, spent = await simulate(
177
+ scenario,
178
+ contract,
179
+ world,
180
+ model=model,
181
+ simulator_prompt=load_simulator_prompt(world_root),
182
+ )
183
+ transcript = from_alk(report, world, spent)
184
+ for exchange in transcript.exchanges:
185
+ if on_exchange:
186
+ on_exchange(exchange)
187
+ else:
188
+ agent = resolve(target)(contract, world, model=model)
189
+ transcript = await converse(
190
+ agent,
191
+ scenario,
192
+ contract,
193
+ world_root=world_root,
194
+ model=model,
195
+ on_exchange=on_exchange,
196
+ )
197
+ # Settled by code first. The judge is only handed the sub-goals whose catalogue entry
198
+ # says nothing observable decides them.
199
+ settled = grade_sub_goals(world, scenario, catalogue, transcript.calls)
200
+ ending = ", ".join(
201
+ f"{name}: {len(rows)} rows"
202
+ for name, rows in sorted(world.observe().state.items())
203
+ )
204
+ judgements, judged_cost = await judge(
205
+ scenario, transcript, contract, catalogue, model=model, ending=ending
206
+ )
207
+ judgements += judge_suite_evals(
208
+ catalogue.suite_evals, scenario, transcript, contract, ending=ending
209
+ )
210
+ return Result(
211
+ scenario=scenario.name,
212
+ tests=scenario.tests,
213
+ problems=[
214
+ f"{name} is not in this catalogue, so nothing graded it"
215
+ for name in ungraded_sub_goals(scenario, catalogue)
216
+ ],
217
+ state_failures=[
218
+ f"{one.name}: {one.said}" for one in settled if not one.held
219
+ ],
220
+ conduct=judgements,
221
+ checkpoints=checkpoints(settled, judgements),
222
+ crashes=[f"{call.name}: {call.error}" for call in transcript.crashed()],
223
+ ended=transcript.ended,
224
+ turns=len(transcript.exchanges),
225
+ calls=len(transcript.calls),
226
+ spent_usd=transcript.spent_usd + judged_cost,
227
+ transcript=transcript.spoken(),
228
+ exchanges=[
229
+ {"speaker": turn.speaker, "text": turn.text}
230
+ for turn in transcript.exchanges
231
+ ],
232
+ actions=transcript.actions(),
233
+ )
234
+ finally:
235
+ world.close()
236
+
237
+
238
+ async def run_suite(
239
+ scenarios: Sequence[Scenario],
240
+ contract: AgentContract,
241
+ world_root: Path,
242
+ *,
243
+ target: str = "local",
244
+ model: str | None = None,
245
+ through_alk: bool = False,
246
+ out: Path | None = None,
247
+ on_result: Callable[[Result], Any] | None = None,
248
+ on_exchange: Callable[[Exchange], Any] | None = None,
249
+ ) -> list[Result]:
250
+ """Run every scenario and write the results out. One failing scenario never stops the rest."""
251
+ destination = Path(out or world_root)
252
+ results: list[Result] = []
253
+ for scenario in scenarios:
254
+ try:
255
+ result = await run_scenario(
256
+ scenario,
257
+ contract,
258
+ world_root,
259
+ target=target,
260
+ model=model,
261
+ through_alk=through_alk,
262
+ on_exchange=on_exchange,
263
+ )
264
+ except Exception as failed:
265
+ # A scenario that could not be run is recorded as unrunnable rather than as a
266
+ # failure of the agent, and the rest of the suite still runs.
267
+ result = Result(
268
+ scenario=scenario.name,
269
+ tests=scenario.tests,
270
+ crashes=[f"could not run: {type(failed).__name__}: {failed}"],
271
+ ended="not-run",
272
+ )
273
+ results.append(result)
274
+ if on_result:
275
+ on_result(result)
276
+
277
+ destination.mkdir(parents=True, exist_ok=True)
278
+ # Records for scenarios this suite did not run are kept, not clobbered. A live call and a
279
+ # local run write to the same file, and re-running two scenarios must not erase the third.
280
+ ran = {result.scenario for result in results}
281
+ kept = [
282
+ record
283
+ for record in load_results(destination)
284
+ if isinstance(record, dict) and record.get("scenario") not in ran
285
+ ]
286
+ merged = kept + json.loads(as_json(results))
287
+ (destination / RUNS).write_text(
288
+ json.dumps(merged, indent=2, ensure_ascii=False), encoding="utf-8"
289
+ )
290
+ (destination / REPORT).write_text(summarise(results), encoding="utf-8")
291
+ return results
292
+
293
+
294
+ def load_results(destination: Path) -> list[dict[str, Any]]:
295
+ path = Path(destination) / RUNS
296
+ return json.loads(path.read_text(encoding="utf-8")) if path.exists() else []
@@ -0,0 +1,184 @@
1
+ """Running a generated world through ALK's own simulation, rather than beside it.
2
+
3
+ The whole reason a generated world subclasses ``EnvironmentAdapter`` is so the runners that
4
+ already exist can drive it. ``ChatEnvironment`` takes ``environment=<adapter>`` and owns the
5
+ synthetic user, the turn loop, the transcript and the report; the browser and voice paths take
6
+ the same adapter. A second loop written here would work for exactly one modality and would have
7
+ to be rewritten for the next one, which is the thing this design exists to avoid.
8
+
9
+ So the split is:
10
+
11
+ - **ALK** drives the simulation: who the customer is, when they speak, when it ends.
12
+ - **The world** answers every tool call, through ``handle_tool_call``.
13
+ - **The harness** grades afterwards, from the state the world is left in plus the transcript.
14
+
15
+ What is written here is only the two adapters between the shapes: a scenario becomes an ALK
16
+ ``Persona``, and the agent under test becomes an ``AgentWrapper``.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from typing import Any
22
+
23
+ from fi.simulate import Persona, Scenario as AlkScenario
24
+ from fi.simulate.agent.wrapper import AgentInput, AgentResponse, AgentWrapper
25
+ from fi.simulate.environments.chat import ChatEnvironment
26
+
27
+ from ..contract import AgentContract
28
+ from ..scenario import Scenario
29
+ from ..world.runtime import GeneratedWorld
30
+ from .targets import LocalAgent
31
+
32
+
33
+ def as_persona(scenario: Scenario, simulator_prompt: str = "") -> Persona:
34
+ """One of our scenarios, in the shape ALK's simulation consumes.
35
+
36
+ ``situation`` is the simulator prompt the harness wrote for this agent with the scenario's
37
+ values filled in. ALK wraps it in its own voice-execution rules, so what goes here is only
38
+ what changes per scenario, not a second set of instructions about how to behave on a call.
39
+
40
+ The structured persona is preserved so LiveKit can vary the caller's identity, speech style,
41
+ and scenario metadata instead of flattening every test into the same generic customer.
42
+ """
43
+ from ..simulator import fill
44
+
45
+ # The LiveKit voice simulator treats ``situation`` as private context rather than a
46
+ # line to recite. Preserve the harness-authored caller rules here: they explicitly keep
47
+ # the simulator in the customer role and stop it from volunteering held-back details.
48
+ # Fall back to the plain scenario instruction for older sessions without a prompt.
49
+ filled = fill(simulator_prompt, scenario.slots())[0] if simulator_prompt else ""
50
+ persona = (
51
+ scenario.persona.model_dump(exclude_none=True)
52
+ if scenario.persona is not None
53
+ else {"name": "customer"}
54
+ )
55
+ return Persona(
56
+ persona=persona,
57
+ situation=filled or scenario.instruction,
58
+ # Said in the person's own terms, because it is read aloud with the situation.
59
+ # ``scenario.tests`` describes what the suite is checking — "agent correctly counts
60
+ # customers filtered by country" — and a person who opens by announcing what the agent
61
+ # is being graded on has told it the answer.
62
+ outcome="get what you came for, or accept that you cannot",
63
+ )
64
+
65
+
66
+ def as_alk_scenario(
67
+ scenarios: list[Scenario], name: str = "harness", simulator_prompt: str = ""
68
+ ) -> AlkScenario:
69
+ return AlkScenario(
70
+ name=name,
71
+ description="generated by the harness",
72
+ dataset=[as_persona(one, simulator_prompt) for one in scenarios],
73
+ )
74
+
75
+
76
+ def _spoken(input: AgentInput) -> str:
77
+ """What the customer just said, as text.
78
+
79
+ ALK passes a message as a mapping, not a string, and hands the whole history alongside it.
80
+ Passing the mapping straight to a session that expects text fails inside the SDK with a
81
+ redacted TypeError, which says nothing about where it came from.
82
+ """
83
+ latest = input.new_message
84
+ if isinstance(latest, dict):
85
+ content = latest.get("content")
86
+ if isinstance(content, list):
87
+ content = " ".join(
88
+ part.get("text", "") for part in content if isinstance(part, dict)
89
+ )
90
+ if content:
91
+ return str(content)
92
+ if isinstance(latest, str) and latest:
93
+ return latest
94
+ for message in reversed(input.messages or []):
95
+ if isinstance(message, dict) and message.get("role") != "assistant":
96
+ content = message.get("content")
97
+ if content:
98
+ return str(content)
99
+ return "(the customer said nothing)"
100
+
101
+
102
+ class ContractAgent(AgentWrapper):
103
+ """The agent under test, in the shape ALK drives agents by.
104
+
105
+ It holds the same session ``LocalAgent`` uses, so the agent being graded is identical either
106
+ way; what changes is who runs the conversation around it. The tool calls it made are reported
107
+ back to ALK so they appear in the transcript, having already gone through the world.
108
+ """
109
+
110
+ def __init__(
111
+ self,
112
+ contract: AgentContract,
113
+ world: GeneratedWorld,
114
+ *,
115
+ model: str | None = None,
116
+ ) -> None:
117
+ self.agent = LocalAgent(contract, world, model=model)
118
+ self.world = world
119
+ self._open = False
120
+
121
+ async def call(self, input: AgentInput) -> AgentResponse:
122
+ if not self._open:
123
+ await self.agent.open()
124
+ self._open = True
125
+
126
+ before = len(self.world.calls)
127
+ said = await self.agent.say(_spoken(input))
128
+ made = self.world.calls[before:]
129
+
130
+ return AgentResponse(
131
+ content=said,
132
+ tool_calls=[
133
+ {"name": call.name, "arguments": call.arguments} for call in made
134
+ ],
135
+ tool_responses=[
136
+ {
137
+ "name": call.name,
138
+ "content": call.error if not call.ok else str(call.result),
139
+ "success": call.ok,
140
+ }
141
+ for call in made
142
+ ],
143
+ )
144
+
145
+ async def aclose(self) -> None:
146
+ if self._open:
147
+ await self.agent.close()
148
+ self._open = False
149
+
150
+ @property
151
+ def spent_usd(self) -> float:
152
+ return self.agent.spent_usd
153
+
154
+
155
+ async def simulate(
156
+ scenario: Scenario,
157
+ contract: AgentContract,
158
+ world: GeneratedWorld,
159
+ *,
160
+ model: str | None = None,
161
+ simulator_prompt: str = "",
162
+ ) -> tuple[Any, float]:
163
+ """Run one scenario through ALK's chat simulation against this world.
164
+
165
+ ``auto_execute_tools`` is off because the agent's tools are bound to the world already and
166
+ have run by the time it answers. Turning it on would execute every call a second time, which
167
+ for a world that really writes rows means every order placed twice.
168
+ """
169
+ agent = ContractAgent(contract, world, model=model)
170
+ try:
171
+ report = await ChatEnvironment().run(
172
+ scenario=as_alk_scenario(
173
+ [scenario], name=scenario.name, simulator_prompt=simulator_prompt
174
+ ),
175
+ agent_callback=agent,
176
+ environment=world,
177
+ auto_execute_tools=False,
178
+ max_turns=max(2, scenario.max_turns),
179
+ min_turns=2,
180
+ modality=contract.modality or "text",
181
+ )
182
+ finally:
183
+ await agent.aclose()
184
+ return report, agent.spent_usd
@@ -0,0 +1,162 @@
1
+ """One scenario, against the real hosted agent, end to end.
2
+
3
+ Everything the harness built is wired together here and then ALK's own voice case places the
4
+ call. The harness does not reimplement any of that: it supplies the world the agent's tools act
5
+ on, the caller's instruction, and the grading afterwards.
6
+
7
+ world + setup ──► webhook ──► public url ──► assistant's own tools repointed
8
+ │
9
+ ALK's voice case places the call ──┘
10
+ │
11
+ the world afterwards + the calls ──► sub-goal checks
12
+
13
+ Run it:
14
+
15
+ set -a; . ./.env.acceptance; set +a
16
+ uv run python -m harness.run.call --name drive_thru --scenario orders_a_big_mac
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import argparse
22
+ import json
23
+ import os
24
+ import subprocess
25
+ import sys
26
+ from pathlib import Path
27
+
28
+ from ..config import artifact_dir
29
+ from ..scenario_tools import load_scenarios
30
+ from .live import grade, wire
31
+
32
+ CASE = os.environ.get("HARNESS_VOICE_CASE", "2.1.2")
33
+
34
+
35
+ # The default runner is part of the harness package and uses ALK's public
36
+ # SimulationSpec/SimulationRunner API. It is overridable only for provider
37
+ # acceptance work; normal customer runs need no second checkout or script.
38
+ VOICE_RUNNER = Path(__file__).with_name("sdk_voice.py")
39
+
40
+
41
+ LIVE_EVENT = "HARNESS_EXCHANGE "
42
+
43
+
44
+ def place_the_call(case: str, dry_run: bool = False, on_exchange=None) -> int:
45
+ """Hand over to ALK's voice case, which owns everything about placing a call."""
46
+ named = os.environ.get("HARNESS_VOICE_RUNNER", "").strip()
47
+ runner = Path(named) if named else VOICE_RUNNER
48
+ if not runner.exists():
49
+ raise RuntimeError(
50
+ f"no voice runner at {runner}. It ships in the harness package; set "
51
+ "HARNESS_VOICE_RUNNER if it lives somewhere else."
52
+ )
53
+ command = [sys.executable, str(runner), case] + (["--dry-run"] if dry_run else [])
54
+ if on_exchange is None:
55
+ return subprocess.call(command)
56
+ process = subprocess.Popen(
57
+ command,
58
+ stdout=subprocess.PIPE,
59
+ stderr=subprocess.STDOUT,
60
+ text=True,
61
+ bufsize=1,
62
+ )
63
+ assert process.stdout is not None
64
+ for line in process.stdout:
65
+ if line.startswith(LIVE_EVENT):
66
+ try:
67
+ on_exchange(json.loads(line[len(LIVE_EVENT) :]))
68
+ except (json.JSONDecodeError, TypeError):
69
+ pass
70
+ else:
71
+ print(line, end="", flush=True)
72
+ return process.wait()
73
+
74
+
75
+ def main(argv: list[str] | None = None) -> int:
76
+ parser = argparse.ArgumentParser(prog="agent-harness-call", description=__doc__)
77
+ parser.add_argument("--name", required=True, help="which agent")
78
+ parser.add_argument("--scenario", required=True, help="which scenario, by name")
79
+ parser.add_argument("--case", default=CASE, help="ALK voice case id")
80
+ parser.add_argument(
81
+ "--dry-run",
82
+ action="store_true",
83
+ help="wire everything up but do not place the call",
84
+ )
85
+ args = parser.parse_args(argv)
86
+
87
+ root = artifact_dir(args.name)
88
+ written = load_scenarios(root)
89
+ scenario = next((one for one in written if one.name == args.scenario), None)
90
+ if scenario is None:
91
+ print(
92
+ f"no scenario called {args.scenario!r}. There is: "
93
+ + ", ".join(one.name for one in written),
94
+ file=sys.stderr,
95
+ )
96
+ return 1
97
+
98
+ world, instruction, webhook, tunnel, url, moved = wire(scenario, root)
99
+ try:
100
+ print(f"agent: {args.name}")
101
+ print(f"scenario: {scenario.name}")
102
+ print(f"webhook: {url}/tool")
103
+ print(f"repointed: {', '.join(moved)}")
104
+ print(f"sub-goals: {', '.join(scenario.sub_goals)}\n")
105
+
106
+ # The caller's instruction reaches the voice case through the environment, so nothing
107
+ # about how a simulated caller behaves is decided twice.
108
+ os.environ["HARNESS_INSTRUCTION"] = instruction
109
+ os.environ["HARNESS_SCENARIO"] = scenario.name
110
+ # The caller prompt is a template the harness fills, never prose it composes, so
111
+ # the generated template travels to the call with everything else.
112
+ # The caller is never handed the grader's pass question. `tests` is written about
113
+ # the agent in the third person, so as an objective it reads as a rubric rather
114
+ # than a motive. What this person wants is already in the instruction.
115
+ os.environ.pop("HARNESS_OUTCOME", None)
116
+ os.environ["HARNESS_PERSONA"] = json.dumps(
117
+ scenario.persona.model_dump(exclude_none=True)
118
+ if scenario.persona is not None
119
+ else {"name": "customer"}
120
+ )
121
+ os.environ["HARNESS_INITIAL_MESSAGE"] = (
122
+ scenario.persona.initial_message if scenario.persona is not None else ""
123
+ )
124
+ os.environ["HARNESS_FIXTURE"] = json.dumps(
125
+ scenario.fixture, ensure_ascii=False, default=str
126
+ )
127
+ os.environ["HARNESS_SCRIPTED_CALLER"] = json.dumps(
128
+ scenario.persona.scripted_caller
129
+ if scenario.persona is not None
130
+ and scenario.persona.scripted_caller is not None
131
+ else {}
132
+ )
133
+
134
+ code = place_the_call(args.case, dry_run=args.dry_run)
135
+ if args.dry_run:
136
+ print("\ndry run: nothing was called, and the world is untouched.")
137
+ return code
138
+
139
+ result = grade(scenario, world, root)
140
+ print()
141
+ print(result.line())
142
+ for one in result.settled:
143
+ print(one.line())
144
+ for name in result.judged:
145
+ print(f" [?] {name} — judged, not graded here")
146
+ print("\nwhat the agent actually did:")
147
+ for call in result.calls or ["(no tool calls reached the world)"]:
148
+ print(f" {call}")
149
+ return 0 if result.settled and result.met == len(result.settled) else 2
150
+ finally:
151
+ webhook.stop()
152
+ if tunnel is not None:
153
+ tunnel.terminate()
154
+ if (root / "environment.json").exists():
155
+ from ..provision import stop_runtime
156
+
157
+ stop_runtime(root)
158
+ world.close()
159
+
160
+
161
+ if __name__ == "__main__":
162
+ raise SystemExit(main())