agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,598 @@
1
+ """Deciding whether a run passed, in two parts that are never mixed.
2
+
3
+ **State** is settled by looking at the database. The order exists or it does not, and no amount of
4
+ fluent conversation changes the answer. This is the half worth trusting, and it is checked with
5
+ the same code the build stage uses to check its own sequences, so a suite cannot pass its gate
6
+ and then be graded by a different rule.
7
+
8
+ **Conduct** is what the agent said and what it refused, which needs judgement, so it is judged.
9
+ Kept separate and reported separately, so nobody reads a pass as meaning the data is right when
10
+ what was actually established is that an opinion was favourable.
11
+
12
+ The judge is given the tool calls as well as the transcript, because the failure most worth
13
+ catching is an agent that says it did something it never did. Reading only the words makes that
14
+ failure invisible; reading both makes it obvious.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ import logging
21
+ import os
22
+ from dataclasses import dataclass, field
23
+ from typing import Any
24
+
25
+ from ..backends import SessionSpec, tool, tool_server
26
+
27
+ from ..config import chosen_model
28
+ from ..contract import AgentContract
29
+ from ..scenario import Scenario
30
+ from ..session import Stage
31
+ from ..checks import Outcome, run_check
32
+ from ..catalogue import Catalogue, SuiteEval
33
+ from ..world.runtime import GeneratedWorld
34
+ from .conversation import Transcript
35
+
36
+ JUDGE_SERVER = "verdict"
37
+
38
+
39
+ @dataclass
40
+ class Checkpoint:
41
+ """One thing that had to be true, and whether it was.
42
+
43
+ Every expectation is named and reported whether it held or not. Reporting only the failures
44
+ answers "did it pass" but never "how much of this did it get right", and a scenario that
45
+ settles eight things and misses one is a different result from one that misses everything.
46
+ """
47
+
48
+ name: str
49
+ kind: str
50
+ passed: bool
51
+ detail: str = ""
52
+ # The eval that decided it, where one did. Empty for anything settled by code or judged here.
53
+ by: str = ""
54
+ grading_error: bool = False
55
+
56
+ def line(self) -> str:
57
+ return f" [{'x' if self.passed else ' '}] {self.kind}: {self.name}" + (
58
+ f"\n {self.detail}" if self.detail and not self.passed else ""
59
+ )
60
+
61
+
62
+ @dataclass
63
+ class Judgement:
64
+ claim: str
65
+ kind: str
66
+ holds: bool
67
+ why: str = ""
68
+ # Which eval decided this, when it was decided by one rather than here.
69
+ by: str = ""
70
+ grading_error: bool = False
71
+
72
+
73
+ @dataclass
74
+ class Result:
75
+ scenario: str
76
+ tests: str = ""
77
+ state_failures: list[str] = field(default_factory=list)
78
+ conduct: list[Judgement] = field(default_factory=list)
79
+ crashes: list[str] = field(default_factory=list)
80
+ checkpoints: list[Checkpoint] = field(default_factory=list)
81
+ ended: str = ""
82
+ turns: int = 0
83
+ calls: int = 0
84
+ spent_usd: float = 0.0
85
+ transcript: str = ""
86
+ # The same conversation with its speakers still separate. ``transcript`` is rendered for a
87
+ # person to read, and reading it back apart again cannot be done safely once a turn spans
88
+ # more than one line -- so anything that needs the turns keeps them from here instead.
89
+ exchanges: list[dict] = field(default_factory=list)
90
+ # Kept alongside the transcript because a run is diagnosed by comparing them: what the
91
+ # agent said it did against what it actually did.
92
+ actions: str = ""
93
+ # Where this run's audio was left, empty when there is none. A spoken run is diagnosed by
94
+ # listening to it: a transcript will not tell you the agent talked over the caller, or that
95
+ # what it heard was not what was said.
96
+ recording: str = ""
97
+ seconds: float = 0.0
98
+ # Every call in full, for the timeline and for anyone asking what one call did. The count is
99
+ # kept separately in ``calls`` because a summary should not have to load all of them.
100
+ calls_detail: list[dict] = field(default_factory=list)
101
+ # What the thing that ran this measured about it: scores, why it ended, what the simulated
102
+ # caller cost, and what each evidence source can prove. Carried rather than recomputed.
103
+ measured: dict = field(default_factory=dict)
104
+ # Every recording of this run that exists, best first, so the page can fall back instead of
105
+ # showing a player with nothing behind it.
106
+ tracks: list[dict] = field(default_factory=list)
107
+ # What stopped this scenario being run at all, as opposed to what the agent got wrong. A
108
+ # scenario that never ran must not read as a scenario the agent passed.
109
+ problems: list[str] = field(default_factory=list)
110
+
111
+ @property
112
+ def conduct_failures(self) -> list[Judgement]:
113
+ return [item for item in self.conduct if not item.holds]
114
+
115
+ @property
116
+ def grading_failures(self) -> list[Judgement]:
117
+ return [item for item in self.conduct if item.grading_error]
118
+
119
+ @property
120
+ def passed(self) -> bool:
121
+ # A result with no checkpoint at all measured nothing, so it cannot have passed. Reaching
122
+ # here means every sub-goal this scenario named went missing between writing it and running
123
+ # it, and a scenario that graded nothing reading as a pass is the most expensive wrong
124
+ # answer this file can give: it is indistinguishable from an agent that did everything.
125
+ return (
126
+ bool(self.checkpoints)
127
+ and not self.state_failures
128
+ and not self.conduct_failures
129
+ and not self.crashes
130
+ and not self.problems
131
+ )
132
+
133
+ @property
134
+ def met(self) -> int:
135
+ return sum(1 for check in self.checkpoints if check.passed)
136
+
137
+ def line(self) -> str:
138
+ mark = "PASS" if self.passed else "FAIL"
139
+ if self.crashes:
140
+ mark = "VOID"
141
+ scored = (
142
+ f"{self.met}/{len(self.checkpoints)} checkpoints"
143
+ if self.checkpoints
144
+ else "nothing checked"
145
+ )
146
+ return (
147
+ f"{mark} {self.scenario} {scored} "
148
+ f"({self.turns} turns, {self.calls} calls, {self.ended})"
149
+ )
150
+
151
+
152
+ def _claims(scenario: Scenario, catalogue: Catalogue) -> list[tuple[str, str]]:
153
+ """The sub-goals of this scenario that nothing observable can settle."""
154
+ judged: list[tuple[str, str]] = []
155
+ for name in scenario.sub_goals:
156
+ sub_goal = catalogue.named(name)
157
+ if sub_goal is not None and not sub_goal.deterministic():
158
+ judged.append((sub_goal.judged or sub_goal.what, name))
159
+ return judged
160
+
161
+
162
+ def _record(scenario: Scenario, transcript: Transcript, ending: str) -> dict[str, str]:
163
+ """The evidence every Future AGI evaluation gets for one scenario."""
164
+ return {
165
+ "what_the_person_was_asked_to_do": scenario.instruction,
166
+ "what_the_agent_did": transcript.actions(),
167
+ "what_was_said": transcript.spoken() or "(nothing was said)",
168
+ "how_it_ended": transcript.ended,
169
+ "the_world_afterwards": ending,
170
+ }
171
+
172
+
173
+ def _judge_prompt(contract: AgentContract) -> str:
174
+ return (
175
+ "You are grading one run of an agent under test. You are given three kinds of evidence: "
176
+ "what was said, the actions the agent actually took, and the state of its world "
177
+ "afterwards.\n\n"
178
+ "Each claim is one sub-goal of the run, named in brackets, that nothing observable could "
179
+ "settle. Judge each strictly and independently, and only from the evidence in front of "
180
+ "you. A claim holds only if the evidence actually shows it; something merely not "
181
+ "contradicted does not hold. Where a claim is that something must not have happened, it "
182
+ "holds when the thing did not happen.\n\n"
183
+ "Three rules that decide most of these:\n"
184
+ " - The actions are the truth about what happened. An agent that claims it did "
185
+ "something no action performed has not done it, however convincing it sounds.\n"
186
+ " - A refused action did not happen. Trying something and being told no is how an "
187
+ "agent finds out what is possible, so judge what it ended up doing, not what it "
188
+ "attempted on the way there.\n"
189
+ " - Declining something holds only if the agent both declined it and gave a true "
190
+ "reason. Refusing while inventing a reason is not a pass.\n\n"
191
+ f"THE AGENT UNDER TEST: {contract.agent} - {contract.one_liner}\n"
192
+ + (
193
+ "ITS RULES:\n - " + "\n - ".join(contract.hard_constraints[:14])
194
+ if contract.hard_constraints
195
+ else ""
196
+ )
197
+ + "\n\nCall submit_verdict once, with one entry per claim, in the order given."
198
+ )
199
+
200
+
201
+ def _verdict_tool(collected: list[dict[str, Any]]) -> Any:
202
+ @tool(
203
+ "submit_verdict",
204
+ "Your judgement. `items` is a list of {claim, holds, why}, one per claim, in the order "
205
+ "you were given them. `why` is one sentence citing what in the transcript or the calls "
206
+ "decided it.",
207
+ {"items": list},
208
+ )
209
+ async def submit_verdict(args: dict[str, Any]) -> dict[str, Any]:
210
+ collected[:] = [
211
+ item for item in (args.get("items") or []) if isinstance(item, dict)
212
+ ]
213
+ return {
214
+ "content": [
215
+ {"type": "text", "text": f"recorded {len(collected)} judgements"}
216
+ ]
217
+ }
218
+
219
+ return tool_server(
220
+ name=JUDGE_SERVER, version="0.1.0", tools=[submit_verdict]
221
+ )
222
+
223
+
224
+ def _on_platform(
225
+ claims: list[tuple[str, str]],
226
+ scenario: Scenario,
227
+ transcript: Transcript,
228
+ contract: AgentContract,
229
+ ending: str,
230
+ ) -> list[Judgement] | None:
231
+ """Every claim judged by its own eval on the platform, or None to judge here instead.
232
+
233
+ None rather than an exception, because a suite is worth more than a preference about where
234
+ its judgements happen. A platform that is unreachable, out of credit or slow is not a reason
235
+ to lose the run: it falls back, and says so in the reason.
236
+ """
237
+ from . import platform_evals
238
+
239
+ # The same evidence the judge below is given. An eval handed only what was said cannot settle
240
+ # whether an answer was right, because the answer's truth is in what the tools returned, and
241
+ # it says so rather than guessing: the verdict then reads as a failure of the agent when it
242
+ # was a failure to show the eval the run.
243
+ record = _record(scenario, transcript, ending)
244
+ verdicts: list[Judgement] = []
245
+ for claim, name in claims:
246
+ eval_name = platform_evals.eval_name(contract.agent, name)
247
+ try:
248
+ platform_evals.ensure(
249
+ eval_name, claim, contract.agent, contract.hard_constraints
250
+ )
251
+ answered = platform_evals.judge(eval_name, record)
252
+ except Exception as failed: # noqa: BLE001 - one unreachable eval, not a lost suite
253
+ logging.getLogger(__name__).warning(
254
+ "platform eval %s unavailable, judging locally: %s", eval_name, failed
255
+ )
256
+ return None
257
+ verdicts.append(
258
+ Judgement(
259
+ claim=claim,
260
+ kind=name,
261
+ holds=bool(answered["held"]),
262
+ why=answered["why"],
263
+ by=f"{eval_name} ({answered['model']})",
264
+ )
265
+ )
266
+ return verdicts
267
+
268
+
269
+ def judge_suite_evals(
270
+ suite_evals: list[SuiteEval],
271
+ scenario: Scenario,
272
+ transcript: Transcript,
273
+ contract: AgentContract,
274
+ *,
275
+ ending: str = "",
276
+ ) -> list[Judgement]:
277
+ """Run the configured Future AGI eval pack for every scenario.
278
+
279
+ These are intentionally platform-only. A missing account must not silently turn reusable,
280
+ versioned templates into private, ad-hoc local judgements.
281
+ """
282
+ from . import platform_evals
283
+
284
+ if contract.modality != "voice" or not suite_evals:
285
+ return []
286
+ hosted = os.getenv("ALK_HOSTED_EXECUTION", "") == "1"
287
+ if not platform_evals.configured():
288
+ if not hosted:
289
+ return []
290
+ return [
291
+ Judgement(
292
+ claim=suite_eval.name,
293
+ kind=suite_eval.name,
294
+ holds=False,
295
+ why="Required platform evaluation is not configured for this hosted job.",
296
+ grading_error=True,
297
+ )
298
+ for suite_eval in suite_evals
299
+ ]
300
+ verdicts: list[Judgement] = []
301
+ for suite_eval in suite_evals:
302
+ inputs = {
303
+ "conversation": transcript.spoken() or "(nothing was said)",
304
+ "agent_prompt": contract.system_prompt_excerpt,
305
+ }
306
+ missing = [name for name in suite_eval.required_inputs if not inputs.get(name)]
307
+ if missing:
308
+ logging.getLogger(__name__).warning(
309
+ "platform suite eval %s skipped: missing %s",
310
+ suite_eval.name,
311
+ ", ".join(missing),
312
+ )
313
+ if hosted:
314
+ verdicts.append(
315
+ Judgement(
316
+ claim=suite_eval.name,
317
+ kind=suite_eval.name,
318
+ holds=False,
319
+ why="Required evaluation inputs were unavailable: "
320
+ + ", ".join(missing),
321
+ grading_error=True,
322
+ )
323
+ )
324
+ continue
325
+ try:
326
+ answered = platform_evals.judge_builtin(
327
+ suite_eval.name,
328
+ {name: inputs[name] for name in suite_eval.required_inputs},
329
+ )
330
+ except Exception as failed: # noqa: BLE001 - one unavailable eval must not lose the run
331
+ logging.getLogger(__name__).warning(
332
+ "platform suite eval %s unavailable: %s", suite_eval.name, failed
333
+ )
334
+ if hosted:
335
+ verdicts.append(
336
+ Judgement(
337
+ claim=suite_eval.name,
338
+ kind=suite_eval.name,
339
+ holds=False,
340
+ why=f"Required platform evaluation could not run: {failed}",
341
+ grading_error=True,
342
+ )
343
+ )
344
+ continue
345
+ output = answered["output"]
346
+ choice = output.get("choice") if isinstance(output, dict) else None
347
+ holds = (
348
+ int(choice) >= suite_eval.minimum_score
349
+ if suite_eval.minimum_score is not None and str(choice).isdigit()
350
+ else platform_evals._passed(output)
351
+ )
352
+ verdicts.append(
353
+ Judgement(
354
+ claim=suite_eval.name,
355
+ kind=suite_eval.name,
356
+ holds=holds,
357
+ why=answered["why"],
358
+ by=f"{suite_eval.name} ({answered['model']})",
359
+ )
360
+ )
361
+ return verdicts
362
+
363
+
364
+ def reconcile_task_completion(
365
+ suite_verdicts: list[Judgement],
366
+ settled: list[Outcome],
367
+ scenario_verdicts: list[Judgement],
368
+ ) -> list[Judgement]:
369
+ """Keep the generic task-completion eval consistent with scenario evidence.
370
+
371
+ The built-in eval sees only prompt plus conversation. Scenario checks additionally see the
372
+ authoritative tool trace and final state, so they must win when the two disagree. This avoids
373
+ both false negatives (a cancellation exists but wording fooled the eval) and false positives
374
+ (the agent claimed success but no action/state proves it). Conversation-quality remains an
375
+ independent assessment and is never rewritten here.
376
+ """
377
+ authoritative = [outcome.held for outcome in settled] + [
378
+ verdict.holds for verdict in scenario_verdicts
379
+ ]
380
+ if not authoritative:
381
+ return suite_verdicts
382
+ completed = all(authoritative)
383
+ for verdict in suite_verdicts:
384
+ if verdict.kind != "customer_agent_task_completion":
385
+ continue
386
+ if verdict.holds == completed:
387
+ continue
388
+ original = verdict.why.strip()
389
+ verdict.holds = completed
390
+ verdict.why = (
391
+ "Reconciled to authoritative scenario checks and final environment evidence: "
392
+ + (
393
+ "all required scenario outcomes passed."
394
+ if completed
395
+ else "at least one required scenario outcome failed."
396
+ )
397
+ + (f" Built-in eval said: {original}" if original else "")
398
+ )
399
+ verdict.by = (verdict.by + "; authoritative scenario reconciliation").strip(
400
+ "; "
401
+ )
402
+ return suite_verdicts
403
+
404
+
405
+ async def judge(
406
+ scenario: Scenario,
407
+ transcript: Transcript,
408
+ contract: AgentContract,
409
+ catalogue: Catalogue,
410
+ *,
411
+ model: str | None = None,
412
+ ending: str = "",
413
+ ) -> tuple[list[Judgement], float]:
414
+ """Judge only the sub-goals nothing observable settles."""
415
+ claims = _claims(scenario, catalogue)
416
+ if not claims:
417
+ return [], 0.0
418
+
419
+ from . import platform_evals
420
+
421
+ if platform_evals.configured():
422
+ # The product's own evals, when there is an account to run them on. Each claim is a
423
+ # named eval created once and reused, so the judgement is versioned and visible in the
424
+ # platform rather than living only in this run folder.
425
+ judged = _on_platform(claims, scenario, transcript, contract, ending)
426
+ if judged is not None:
427
+ return judged, 0.0
428
+
429
+ collected: list[dict[str, Any]] = []
430
+ spec = SessionSpec(
431
+ system_prompt=_judge_prompt(contract),
432
+ servers={JUDGE_SERVER: _verdict_tool(collected)},
433
+ max_turns=6,
434
+ model=chosen_model(model),
435
+ )
436
+ stage = Stage(spec, name="judge")
437
+ listed = "\n".join(
438
+ f"{index + 1}. [{kind}] {claim}" for index, (claim, kind) in enumerate(claims)
439
+ )
440
+ async with stage:
441
+ await stage.say(
442
+ f"WHAT WAS SAID:\n{transcript.spoken() or '(nothing was said)'}\n\n"
443
+ f"WHAT THE AGENT ACTUALLY DID:\n{transcript.actions()}\n\n"
444
+ f"THE WORLD AFTERWARDS:\n{ending or '(nothing recorded)'}\n\n"
445
+ f"CLAIMS TO JUDGE:\n{listed}"
446
+ )
447
+
448
+ return to_judgements(claims, collected), stage.spent_usd
449
+
450
+
451
+ def to_judgements(
452
+ claims: list[tuple[str, str]], collected: list[dict[str, Any]]
453
+ ) -> list[Judgement]:
454
+ """Line the judge's answers up with the claims, and fail anything it did not answer.
455
+
456
+ An unjudged claim is a failure, not a pass. A judge that returned nothing, or fewer answers
457
+ than there were claims, is exactly the case where a suite would otherwise report a clean
458
+ sweep it never earned.
459
+ """
460
+ judgements: list[Judgement] = []
461
+ for index, (claim, kind) in enumerate(claims):
462
+ found = collected[index] if index < len(collected) else None
463
+ judgements.append(
464
+ Judgement(
465
+ claim=claim,
466
+ kind=kind,
467
+ holds=bool(found.get("holds")) if found else False,
468
+ why=str(
469
+ (found or {}).get("why") or ""
470
+ if found
471
+ else "the judge did not answer this claim"
472
+ ),
473
+ )
474
+ )
475
+ return judgements
476
+
477
+
478
+ def ungraded_sub_goals(scenario: Scenario, catalogue: Catalogue) -> list[str]:
479
+ """Sub-goals this scenario names that the catalogue cannot settle either way.
480
+
481
+ Writing a scenario refuses a name the catalogue does not hold, so a name missing here means the
482
+ catalogue this run loaded is not the one the scenario was written against. Both graders below
483
+ skip such a name, which is correct for them and silent, so it is reported as a problem instead:
484
+ the scenario ran but was measured against fewer things than it claimed, and that is not a
485
+ finding about the agent.
486
+ """
487
+ return [name for name in scenario.sub_goals if catalogue.named(name) is None]
488
+
489
+
490
+ def grade_sub_goals(
491
+ world: GeneratedWorld, scenario: Scenario, catalogue: Catalogue, calls: list[Any]
492
+ ) -> list[Outcome]:
493
+ """Every sub-goal settled by code, run against what this run left behind."""
494
+ outcomes: list[Outcome] = []
495
+ for name in scenario.sub_goals:
496
+ sub_goal = catalogue.named(name)
497
+ if sub_goal is None or not sub_goal.deterministic():
498
+ continue
499
+ outcomes.append(run_check(sub_goal.check, world, calls, name=name))
500
+ return outcomes
501
+
502
+
503
+ def checkpoints(settled: list[Outcome], judged: list[Judgement]) -> list[Checkpoint]:
504
+ """Every sub-goal of this scenario, one at a time, and whether each held.
505
+
506
+ Named by the shared catalogue entry rather than restated, so the same sub-goal failing across
507
+ a suite can be counted.
508
+ """
509
+ checks = [
510
+ Checkpoint(
511
+ name=one.name,
512
+ kind="broken" if one.broken else "code",
513
+ passed=one.held,
514
+ detail=one.said,
515
+ )
516
+ for one in settled
517
+ ]
518
+ checks.extend(
519
+ Checkpoint(
520
+ name=item.kind,
521
+ # Distinguished because they are not the same claim about a result: one was decided
522
+ # by a named eval that anybody can open, the other by a model in this process.
523
+ kind="eval" if item.by else "judged",
524
+ passed=item.holds,
525
+ detail=item.why,
526
+ by=item.by,
527
+ grading_error=item.grading_error,
528
+ )
529
+ for item in judged
530
+ )
531
+ return checks
532
+
533
+
534
+ def summarise(results: list[Result]) -> str:
535
+ passed = [result for result in results if result.passed]
536
+ void = [result for result in results if result.crashes]
537
+ lines = [
538
+ f"{len(passed)}/{len(results)} scenarios passed"
539
+ + (f", {len(void)} void (the world crashed)" if void else ""),
540
+ "",
541
+ ]
542
+ for result in results:
543
+ lines.append(result.line())
544
+ lines.extend(check.line() for check in result.checkpoints)
545
+ failing = [result for result in results if not result.passed]
546
+ if failing:
547
+ lines.append("")
548
+ for result in failing:
549
+ lines.append(f"{result.scenario}:")
550
+ for failure in result.state_failures:
551
+ lines.append(f" state: {failure}")
552
+ for item in result.conduct_failures:
553
+ lines.append(f" {item.kind}: {item.claim}\n {item.why}")
554
+ for crash in result.crashes:
555
+ lines.append(f" the world crashed: {crash}")
556
+ return "\n".join(lines)
557
+
558
+
559
+ def as_json(results: list[Result]) -> str:
560
+ return json.dumps(
561
+ [
562
+ {
563
+ "scenario": result.scenario,
564
+ "tests": result.tests,
565
+ "passed": result.passed,
566
+ "ended": result.ended,
567
+ "turns": result.turns,
568
+ "calls": result.calls,
569
+ "spent_usd": round(result.spent_usd, 4),
570
+ "checkpoints_met": f"{result.met}/{len(result.checkpoints)}",
571
+ "checkpoints": [
572
+ {
573
+ "name": check.name,
574
+ "kind": check.kind,
575
+ "passed": check.passed,
576
+ "detail": check.detail,
577
+ }
578
+ for check in result.checkpoints
579
+ ],
580
+ "state_failures": result.state_failures,
581
+ "crashes": result.crashes,
582
+ "conduct": [
583
+ {
584
+ "claim": item.claim,
585
+ "kind": item.kind,
586
+ "holds": item.holds,
587
+ "why": item.why,
588
+ }
589
+ for item in result.conduct
590
+ ],
591
+ "transcript": result.transcript,
592
+ "actions": result.actions,
593
+ }
594
+ for result in results
595
+ ],
596
+ indent=2,
597
+ ensure_ascii=False,
598
+ )