agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,915 @@
1
+ """Stage three: write the scenarios the agent will be tested with.
2
+
3
+ Reads the contract and the world that was built from it, and produces scenarios grounded in both.
4
+ The stage can look at the world and run calls against throwaway copies of it, which is what keeps
5
+ a scenario about a real record rather than a plausible-sounding one.
6
+
7
+ Like the other stages it stays open. A suite is usually right on the second look, and "make three
8
+ of these harder" is the next thing said rather than a regeneration from nothing.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import asyncio
14
+ import hashlib
15
+ import logging
16
+ import os
17
+ import random
18
+ from dataclasses import dataclass
19
+ from collections.abc import Awaitable, Callable
20
+ from pathlib import Path
21
+ from typing import Any
22
+
23
+ from .backends import SessionSpec, ToolServer, tool, tool_server
24
+
25
+ from .config import artifact_dir, chosen_model, discovered_skills, load_skill
26
+ from .catalogue import load_catalogue
27
+ from .contract import AgentContract
28
+ from .scenario import Scenario, suite_diversity_problems, voicemail_enabled
29
+ from .scenario_tools import (
30
+ journalled,
31
+ worth_delegating,
32
+ SCENARIO_SERVER,
33
+ load_scenarios,
34
+ scenario_tools,
35
+ world_summary,
36
+ write_scenarios,
37
+ )
38
+ from .session import Stage
39
+ from .tools import schema
40
+
41
+ logger = logging.getLogger(__name__)
42
+
43
+ SKILL = "write-scenarios"
44
+ PLAN_SKILL = "plan-suite"
45
+
46
+ # The review pass runs its own tool server, kept apart from the writers' one so a reviewer can
47
+ # only report gaps and never submit or save a scenario itself.
48
+ REVIEW_SERVER = "suite-review"
49
+
50
+
51
+ # Turns a scenario costs in practice: look at the world, rehearse the calls, submit, and often
52
+ # one more to correct what a gate refused.
53
+ TURNS_EACH = 3
54
+ # Enough to write a handful without the budget being the thing that stops it.
55
+ TURNS_FLOOR = 120
56
+
57
+
58
+ def turns_for(wanted: int) -> int:
59
+ """A turn budget that grows with the suite being asked for.
60
+
61
+ A fixed ceiling is what made asking for a large suite pointless: generation stopped partway
62
+ through, and `save_scenarios` refuses a count that does not match what was asked for, so a run
63
+ that asked for fifty and reached twenty-eight saved nothing at all. The budget has to follow
64
+ the request, or the request cannot be honoured.
65
+ """
66
+ return max(TURNS_FLOOR, wanted * TURNS_EACH + 40)
67
+
68
+
69
+ def open_stage(
70
+ contract: AgentContract,
71
+ *,
72
+ out: Path | None = None,
73
+ wanted: int = 10,
74
+ ask: Callable[..., Any] | None = None,
75
+ max_turns: int = 0,
76
+ ) -> tuple[Stage, Path]:
77
+ """A live write-the-scenarios stage, and where it will write."""
78
+ destination = out or artifact_dir(contract.agent)
79
+ server, kept = scenario_tools(contract, destination, destination, wanted=wanted)
80
+ spec = SessionSpec(
81
+ # Same ordering as the slice writer: the agent and its world before the method.
82
+ system_prompt=(
83
+ f"## This agent\n\n{contract.brief(with_data=True)}"
84
+ f"\n\n## Its world\n\n{world_summary(destination)}"
85
+ f"\n\n{load_skill(SKILL)}"
86
+ # Planning and writing are two stages. This session does the first, so it gets both;
87
+ # a slice writer gets the writing skill alone, because the plan is already made and
88
+ # widening a slice is the one thing it must not do.
89
+ + (f"\n\n{load_skill(PLAN_SKILL)}" if worth_delegating(wanted) else "")
90
+ # Whatever this kind of agent adds on top. A file under skills/kinds/ that
91
+ # declares `applies_to: modality=<kind>` is appended here, so supporting a
92
+ # new kind of agent is adding that file and nothing else.
93
+ # `voicemail` gates the mailbox skill the way `modality` gates this one.
94
+ + discovered_skills(
95
+ modality=contract.modality,
96
+ voicemail="on" if voicemail_enabled() else "off",
97
+ )
98
+ + (
99
+ f"\n\nWrite {wanted} scenarios."
100
+ if not kept
101
+ else f"\n\n{len(kept)} scenarios already exist and are loaded: "
102
+ + ", ".join(scenario.name for scenario in kept)
103
+ + ". Submitting one under an existing name replaces it."
104
+ )
105
+ ),
106
+ servers={SCENARIO_SERVER: server},
107
+ builtins=("AskUserQuestion",),
108
+ cwd=str(destination.parent if destination.parent.exists() else Path.cwd()),
109
+ max_turns=max_turns or turns_for(wanted),
110
+ model=chosen_model(),
111
+ ask=ask,
112
+ thinking=True,
113
+ # Silence means something different once the writing is delegated: see the constant.
114
+ idle_timeout_seconds=(
115
+ QUIET_WHILE_DELEGATING_SECONDS if worth_delegating(wanted) else 0.0
116
+ ),
117
+ )
118
+ return Stage(spec, name=SKILL), destination
119
+
120
+
121
+ def opening(contract: AgentContract, wanted: int = 10, existing: int = 0) -> str:
122
+ if existing:
123
+ return (
124
+ f"There are already {existing} scenarios for {contract.agent!r}, and they are "
125
+ "loaded. Use inspect_scenario before changing each one so every unchanged field is "
126
+ "preserved exactly. Say what you want changed, or add to them. Anything you submit "
127
+ "under an existing name replaces it."
128
+ )
129
+ return (
130
+ f"Write {wanted} scenarios for {contract.agent!r}.\n\n"
131
+ "Look at the world first with inspect_world so every scenario names real records, and "
132
+ "read the sub-goals already defined. After that inspection, immediately work out and "
133
+ "submit one scenario at a time; never hold the whole suite in one long response. Emit a "
134
+ "tool call after each scenario so progress is visible and proved work survives a stop. "
135
+ "Work out each scenario's solution with try_calls before you submit it, because a "
136
+ "scenario is only kept if its solution passes its own "
137
+ "checks and those checks fail without it. In a source-provisioned world, keep each "
138
+ "solution step's arguments exactly model-facing. If the raw dependency needs trusted "
139
+ "fields injected by the worker, put its complete payload in environment_arguments; "
140
+ "never pretend the model supplied an internal identifier, a resolved lookup, a priced "
141
+ "result, or any other value it could not have known. Treat every contract phrase that ties "
142
+ "a value to this conversation literally: the reference "
143
+ "solution must create that state earlier in the same conversation. Never pre-seed "
144
+ "opaque state that the agent has no public tool or session state to retrieve. Cover the "
145
+ "ordinary case, the request that has "
146
+ "to be refused, the rule under pressure, and at least one where state has to carry "
147
+ "across several turns. If a proof says an intended check is vacuous or broken, repair "
148
+ "that named sub-goal with add_sub_goal and resubmit. Never evade a gate by deleting a "
149
+ "check for behavior the scenario still claims to test. Then save_scenarios."
150
+ + (
151
+ "\n\nFor a suite rather than one scenario, say briefly how you are splitting it "
152
+ "across the agent's use cases and then write it with generate_suite in the same "
153
+ "turn: it runs a writer per use case at the same time and saves what they prove."
154
+ if worth_delegating(wanted)
155
+ else ""
156
+ )
157
+ )
158
+
159
+
160
+ def load(destination: Path) -> list[Scenario]:
161
+ """The scenarios written for this agent, if any have been."""
162
+ return load_scenarios(Path(destination))
163
+
164
+
165
+ # What a suite costs, and what it is allowed to cost.
166
+ #
167
+ # Writers run as separate model sessions, so wall clock is roughly the number of scenarios
168
+ # divided by how many run at once. The two ceilings below exist for different reasons: one
169
+ # protects the machine, the other protects the person waiting. Asking for a thousand scenarios
170
+ # is a reasonable thing to want and an unreasonable thing to do in one go, so a large ask is
171
+ # served a batch at a time with the rest offered back.
172
+ AT_ONCE = 4
173
+ # Writers each drive their own model session, so this is a request rate as much as a concurrency.
174
+ # Eight of them exhausts the provider quota and the writers that get the 429 lose their whole
175
+ # slice, which costs more scenarios than the extra concurrency buys.
176
+ MOST_AT_ONCE = int(os.environ.get("HARNESS_WRITERS_AT_ONCE") or 4)
177
+ # How many a single generate_suite pass will write. A hosted run is unattended, so a cap here
178
+ # does not pause for a person, it just returns fewer than were asked for and stops. Kept as a
179
+ # backstop against a runaway ask rather than as a batch size.
180
+ MOST_IN_ONE_GO = 1000
181
+
182
+ # How many times the suite is reviewed and topped up after the first pass. One is enough to
183
+ # catch a slice that came back short or a use case nobody covered; more turns it into a loop
184
+ # that keeps finding smaller things to say.
185
+ TOP_UP_ROUNDS = 1
186
+
187
+ # How long a session may say nothing before the harness treats it as hung, where the default of
188
+ # ten minutes is wrong for this stage.
189
+ #
190
+ # The planning session's whole turn is one `generate_suite` call, and that call does not return
191
+ # until the writers it started have finished. It is working the entire time and has nothing to
192
+ # emit while it works, so the default bound kills a fan-out mid-flight and throws away everything
193
+ # the writers proved. A hundred scenarios across four writers is comfortably an hour, and any
194
+ # writer may also be waiting out a quota refusal inside that.
195
+ #
196
+ # Still bounded, because the reason the bound exists is real: a dropped provider stream leaves a
197
+ # session alive forever. The outer bound is the run's own authoring deadline.
198
+ QUIET_WHILE_DELEGATING_SECONDS = 5400.0
199
+ # A writer is quiet while one of its own tool calls runs, and its longest is a proof: restore the
200
+ # world, apply setup, play the solution, then play it again against an untouched world. Minutes,
201
+ # not an hour, and keeping this shorter than the planner's bound is what frees a stuck writer's
202
+ # slot for the next slice instead of holding it until the whole stage times out.
203
+ QUIET_WHILE_WRITING_SECONDS = 1800.0
204
+
205
+
206
+ # A session refused by the provider is retried rather than abandoned: its work is still worth doing and
207
+ # a slice keeps what it already proved. The quota is measured over a minute, so each wait clears a
208
+ # minute; a shorter one asks inside the same window and is refused again for the same reason. What is
209
+ # bounded is the total, not the number of tries: five minutes of waiting is worth a slice, and a run
210
+ # that waits longer than that is not going to be rescued by waiting more.
211
+ RATE_LIMIT_BACKOFF_SECONDS = 60
212
+ RATE_LIMIT_JITTER_SECONDS = 30
213
+ RATE_LIMIT_TOTAL_WAIT_SECONDS = 300
214
+
215
+
216
+ def _rate_limited(said: str) -> bool:
217
+ """Whether this is the provider refusing for rate or quota rather than for what was asked."""
218
+ lowered = said.lower()
219
+ return any(
220
+ mark in lowered
221
+ for mark in ("429", "resource_exhausted", "resourceexhausted", "rate limit", "quota")
222
+ )
223
+
224
+
225
+ def _refusal_in(broke: BaseException | None, ended: Any) -> str:
226
+ """The refusal, when a turn or an exception is one for rate or quota, and empty otherwise.
227
+
228
+ Two shapes because a backend has two ways of reporting a dead model call, and only one of them
229
+ is an exception. The Vertex backend never raises: it catches everything and finishes the turn
230
+ with `outcome` "failed" and the provider's own words in `error`. A retry that watched only for
231
+ exceptions therefore never fired on the backend the hosted run actually uses, which is how a
232
+ quota refusal went on costing a whole slice while the waiting code looked correct.
233
+ """
234
+ if broke is not None:
235
+ said = f"{type(broke).__name__} {broke}"
236
+ return said if _rate_limited(said) else ""
237
+ if str(getattr(ended, "outcome", "") or "") != "failed":
238
+ return ""
239
+ said = str(getattr(ended, "error", "") or "")
240
+ return said if _rate_limited(said) else ""
241
+
242
+
243
+ def _refusal_pause() -> float:
244
+ """How long to wait before asking again, spread out so sessions do not return together.
245
+
246
+ Sessions are refused at the same moment because they ask at the same moment, so a fixed wait has
247
+ them all wake together and refuse together. Each wait clears the quota's minute and carries
248
+ jitter on top, which is what breaks the lockstep.
249
+ """
250
+ return round(
251
+ RATE_LIMIT_BACKOFF_SECONDS + random.uniform(0, RATE_LIMIT_JITTER_SECONDS), 1
252
+ )
253
+
254
+
255
+ async def survive_refusal(
256
+ run: Callable[[], Awaitable[Any]],
257
+ *,
258
+ what: str,
259
+ on_event: Callable[..., Any] | None = None,
260
+ enough: Callable[[], bool] | None = None,
261
+ ) -> Any:
262
+ """Run something that talks to the model, waiting out a refusal for rate or quota.
263
+
264
+ Used by every session that drives its own model turn, because any of them can be the one the
265
+ provider refuses, and losing the planning turn costs the whole suite rather than one slice.
266
+ Anything that is not a rate or quota refusal is raised, so a real fault still fails fast.
267
+ """
268
+ waited = 0.0
269
+ while True:
270
+ broke: BaseException | None = None
271
+ ended: Any = None
272
+ try:
273
+ ended = await run()
274
+ except Exception as raised: # noqa: BLE001 - classified below, re-raised when not a refusal
275
+ broke = raised
276
+ refusal = _refusal_in(broke, ended)
277
+ pause = _refusal_pause()
278
+ spent_out = waited + pause > RATE_LIMIT_TOTAL_WAIT_SECONDS
279
+ if not refusal or spent_out or (enough is not None and enough()):
280
+ if broke is not None:
281
+ raise broke
282
+ return ended
283
+ waited += pause
284
+ logger.warning(
285
+ "%s was refused for rate or quota; waiting %ss (%ss of %ss spent): %s",
286
+ what, pause, round(waited), RATE_LIMIT_TOTAL_WAIT_SECONDS, refusal[:200],
287
+ )
288
+ if on_event:
289
+ on_event({"type": "waiting_on_provider", "what": what, "seconds": pause})
290
+ await asyncio.sleep(pause)
291
+
292
+
293
+ @dataclass(frozen=True)
294
+ class Slice:
295
+ """One writer's share of a suite: what to write, how much, and why it is worth writing."""
296
+
297
+ use_case: str
298
+ angle: str = ""
299
+ count: int = 1
300
+ why: str = ""
301
+
302
+ def named(self) -> str:
303
+ return f"{self.use_case}: {self.angle}" if self.angle else self.use_case
304
+
305
+
306
+ def even_slices(wanted: int, use_cases: list[str]) -> list[Slice]:
307
+ """The fallback split, when nobody said how the work should be divided.
308
+
309
+ Evenly, with the remainder going to the ones named first, because a contract lists its
310
+ primary use cases before its marginal ones. It is a poor plan and it is meant to be: a use
311
+ case with one real branch gets the same share as one with six, so the first pads and the
312
+ second under-covers. It exists so a caller that supplies no plan still gets a suite.
313
+ """
314
+ if not use_cases:
315
+ return []
316
+ if wanted <= len(use_cases):
317
+ return [Slice(use_case=case, count=1) for case in use_cases[:wanted]]
318
+ each, extra = divmod(wanted, len(use_cases))
319
+ return [
320
+ Slice(use_case=case, count=each + (1 if i < extra else 0))
321
+ for i, case in enumerate(use_cases)
322
+ ]
323
+
324
+
325
+ def planned(wanted: int, use_cases: list[str], given: list[dict] | None) -> list[Slice]:
326
+ """The split this suite will actually be written to.
327
+
328
+ A plan supplied by the caller wins, because whoever is talking to the person has just read
329
+ the contract and the world and knows which use cases have something in them. Sizing every
330
+ use case identically is the thing that made suites pad in one place and under-cover in
331
+ another, and the plan is the only part of the process that knows the difference.
332
+
333
+ Anything the plan leaves out is filled in evenly, and anything it over-asks for is trimmed,
334
+ so a plan can be rough without producing a suite nobody asked for.
335
+ """
336
+ if not given:
337
+ return even_slices(wanted, use_cases)
338
+
339
+ known = {case.strip().lower(): case for case in use_cases}
340
+ slices: list[Slice] = []
341
+ for one in given:
342
+ if not isinstance(one, dict):
343
+ continue
344
+ case = str(one.get("use_case") or "").strip()
345
+ if not case:
346
+ continue
347
+ # Match the contract's own wording where the plan paraphrased it, so a slice is filed
348
+ # under a use case the coverage count recognises rather than a near-miss of one.
349
+ case = known.get(case.lower(), case)
350
+ try:
351
+ count = max(1, int(one.get("count") or 1))
352
+ except (TypeError, ValueError):
353
+ count = 1
354
+ slices.append(
355
+ Slice(
356
+ use_case=case,
357
+ angle=str(one.get("angle") or "").strip(),
358
+ count=count,
359
+ why=str(one.get("why") or "").strip(),
360
+ )
361
+ )
362
+ if not slices:
363
+ return even_slices(wanted, use_cases)
364
+
365
+ # Trim from the end rather than scaling everything down: the plan put its most valuable
366
+ # slices first, and shaving one scenario off each is how a deliberate plan becomes an even
367
+ # one again.
368
+ total = sum(one.count for one in slices)
369
+ while total > wanted and slices:
370
+ last = slices[-1]
371
+ if last.count > 1:
372
+ slices[-1] = Slice(last.use_case, last.angle, last.count - 1, last.why)
373
+ else:
374
+ slices.pop()
375
+ total = sum(one.count for one in slices)
376
+ return slices
377
+
378
+
379
+ # Initial letters dealt out so parallel writers cannot invent the same people. Three per writer, which
380
+ # is enough choice to suit a scenario and keeps seven writers disjoint before the letters wrap; beyond
381
+ # that two writers share initials but still choose different names. Numbers are partitioned by a
382
+ # three-digit prefix instead: a hundred slots collided twice on a run with twenty slices, which is
383
+ # what the birthday arithmetic predicts, and a thousand makes it rare. A repeated verification code is
384
+ # a real collision; a repeated initial is not.
385
+ _NAME_LETTERS = "ABCDEFGHIJKLMNOPRSTVWY"
386
+ _LETTER_BLOCK = "abc"
387
+
388
+
389
+ def _slot(of: str, index: int) -> int:
390
+ """A stable number for one slice, so its share of the value space does not move between passes.
391
+
392
+ Using the position in the current batch looked right and was not: a second `generate_suite` pass
393
+ numbers its slices from zero again, so its first writer is handed the same letters and the same
394
+ leading digits as the first writer of the pass before it, and their codes collide. Measured on a
395
+ 377-scenario run: two verification codes shared, both between passes. Derived from the slice's own
396
+ name instead, which does not change when the batch does, and deterministically so two runs of the
397
+ same plan partition the same way.
398
+ """
399
+ if not of:
400
+ return index
401
+ return int(hashlib.sha256(of.encode("utf-8")).hexdigest()[:8], 16)
402
+
403
+
404
+ def callers_for(index: int, wanted: int, slice_name: str = "") -> str:
405
+ """Which callers this slice should write, so the suite varies across slices as well as within.
406
+
407
+ Instruction alone cannot do this. Each writer is blind to the others, so each independently
408
+ picks the safest value and the suite converges on it: measured across three suites, more
409
+ than half the callers came out "Professional and formal" and over three quarters American,
410
+ with nobody doing anything wrong. Worse, a slice writing a single scenario has nothing to
411
+ vary at all.
412
+
413
+ So the spread is dealt out here, the same way the work is. Each slice is handed a different
414
+ starting point in the platform's own vocabularies and told to begin there. It is a
415
+ suggestion rather than a rule, because the caller still has to suit the scenario: a stolen
416
+ phone is not a cheerful call whatever this hands out.
417
+ """
418
+ from .persona_guides import offered
419
+
420
+ people = offered("personality")
421
+ accents = offered("accent")
422
+ if not people:
423
+ return ""
424
+ picks = [people[(index + step) % len(people)] for step in range(max(1, wanted))]
425
+ said = (
426
+ "\n\nStart from these callers, and move off them only where the scenario calls for "
427
+ f"somebody else: {', '.join(picks)}."
428
+ )
429
+ # Writers cannot see each other, so left to themselves they invent the same handful of people and
430
+ # the same round numbers, and the suite comes back with one name on a dozen scenarios and one
431
+ # verification code shared between them. Partitioning the space of values costs nothing and makes
432
+ # a collision impossible: each writer owns some initial letters and one leading digit, so no
433
+ # shared list of names or codes has to exist for the values to stay distinct.
434
+ slot = _slot(slice_name, index)
435
+ # More letters where the slice is larger: a writer inventing twelve people from three initials
436
+ # reuses a name, which is most of why distinctness measured 73 percent rather than the 90 the
437
+ # suite rule wants.
438
+ block = max(len(_LETTER_BLOCK), min(8, (max(1, wanted) + 2) // 3))
439
+ letters = "".join(
440
+ _NAME_LETTERS[(slot * block + step) % len(_NAME_LETTERS)] for step in range(block)
441
+ )
442
+ said += (
443
+ f"\n\nEvery person you invent must have a given name beginning with one of {letters}, and "
444
+ "every number you invent that the agent will look up, a code or a reference or an account "
445
+ f"number, must begin with {slot % 1000:03d}. Other writers own the other letters and "
446
+ "prefixes, so this is what keeps two scenarios from sharing a name or a code. Within your own "
447
+ "slice, no two people may share a given name and no two scenarios may share a code or a "
448
+ "reference: the prefix keeps you clear of other writers, it does not keep you clear of "
449
+ "yourself."
450
+ )
451
+ if accents:
452
+ # Spread several offered accents across this writer's callers rather than naming just one,
453
+ # so the suite does not collapse to a single default accent and the agent's speech handling
454
+ # is genuinely varied.
455
+ spread = [
456
+ accents[(index + step) % len(accents)]
457
+ for step in range(min(len(accents), max(2, wanted)))
458
+ ]
459
+ said += (
460
+ " Give your callers varied accents from the offered set, a different one per caller "
461
+ f"where it fits rather than defaulting everyone to the same accent: {', '.join(spread)}. "
462
+ "A suite where every caller sounds the same is a missed test of the agent's speech "
463
+ "handling, so do not make them all American unless a scenario truly requires it."
464
+ )
465
+ return said
466
+
467
+
468
+ def brief_for(
469
+ contract: AgentContract, mine: Slice, siblings: list[Slice], callers: str
470
+ ) -> str:
471
+ """What one writer is told: its share, what everyone else holds, and the bar.
472
+
473
+ Written as a brief rather than a template because a writer that cannot see its siblings
474
+ will otherwise write what they are writing. Naming their angles is cheaper than discovering
475
+ the overlap at the merge and throwing the loser away.
476
+ """
477
+ others = "\n".join(f" - {one.named()}" for one in siblings if one is not mine)
478
+ aim = f" {mine.use_case}"
479
+ if mine.angle:
480
+ aim += f"\n Angle: {mine.angle}"
481
+ if mine.why:
482
+ aim += f"\n Worth testing because: {mine.why}"
483
+
484
+ return (
485
+ f"Write {mine.count} scenario{'s' if mine.count != 1 else ''} for {contract.agent!r}, "
486
+ "all of them within this one slice:\n\n"
487
+ f"{aim}\n\n"
488
+ + (
489
+ "The rest of the suite is being written at the same time by others, covering:\n"
490
+ f"{others}\n\nStay out of theirs. A scenario that strays is either a duplicate of "
491
+ "somebody else's or a gap in yours.\n\n"
492
+ if others
493
+ else ""
494
+ )
495
+ + "Every scenario carries this use case verbatim in `use_case`, and its own one-line "
496
+ "`branch` saying what makes it different from the others you write here. Branches are "
497
+ "where the variety lives: the ordinary path, the branch that cannot be completed, the "
498
+ "rule under pressure, state that has to carry across turns, the same request against a "
499
+ "differently seeded world.\n\n"
500
+ "What each one has to be, before you submit it:\n"
501
+ " - every value real, read out of the world with inspect_world, never invented\n"
502
+ " - an instruction that is a circumstance the person is living through, not a script "
503
+ "of lines to say\n"
504
+ " - a setup that makes true whatever the instruction presumes, and a ready check that "
505
+ "proves it\n"
506
+ " - a solution worked out with try_calls first, so the gates are not where you find "
507
+ "out it cannot be passed\n"
508
+ " - sub-goals named from the shared catalogue, and checks that assert the right call "
509
+ "with the right arguments or the right end state, never that something merely happened\n"
510
+ " - a scenario a competent agent could plausibly fail. If any correct implementation "
511
+ "passes it for free, it teaches nothing and is not worth the run\n\n"
512
+ "Look at the world first, and read the sub-goals already defined. Submit each scenario "
513
+ "with submit_scenario and then stop: do not save, and do not ask what to do next. "
514
+ "Whoever asked for this collects the suite and writes it." + callers
515
+ )
516
+
517
+
518
+ async def _write_slice(
519
+ contract: AgentContract,
520
+ mine: Slice,
521
+ siblings: list[Slice],
522
+ *,
523
+ index: int,
524
+ destination: Path,
525
+ on_event: Callable[..., Any] | None,
526
+ ask: Callable[..., Any] | None,
527
+ ) -> list[Scenario]:
528
+ """One slice, written by its own session. Returns what it proved, unsaved."""
529
+ server, kept = scenario_tools(
530
+ contract,
531
+ destination,
532
+ destination,
533
+ wanted=mine.count,
534
+ can_save=False,
535
+ start_from=[],
536
+ )
537
+ logger.info("slice starting: %s (wants %s)", mine.named(), mine.count)
538
+ seen = 0
539
+
540
+ def watch(event: Any) -> None:
541
+ # Report as they land rather than at the end. A slice that proves its first scenario
542
+ # four minutes in is the difference between a run that looks alive and one that does not.
543
+ nonlocal seen
544
+ if len(kept) != seen:
545
+ seen = len(kept)
546
+ logger.info("slice %s proved %s of %s", mine.named(), seen, mine.count)
547
+ if on_event:
548
+ on_event(event)
549
+
550
+ # A slice writer never saves the suite; withholding the tool structurally means no backend
551
+ # has to be told to deny it.
552
+ sliced = SessionSpec(
553
+ # The agent and its world come first, the method second. Grounding evidence read
554
+ # before the instructions that operate on it is followed more closely than the
555
+ # same evidence buried between the instructions and the task.
556
+ system_prompt=(
557
+ f"## This agent\n\n{contract.brief(with_data=True)}"
558
+ f"\n\n## Its world\n\n{world_summary(destination)}"
559
+ f"\n\n{load_skill(SKILL)}"
560
+ # Whatever this kind of agent adds on top. A file under skills/kinds/ that
561
+ # declares `applies_to: modality=<kind>` is appended here, so supporting a
562
+ # new kind of agent is adding that file and nothing else.
563
+ # `voicemail` gates the mailbox skill the way `modality` gates this one.
564
+ + discovered_skills(
565
+ modality=contract.modality,
566
+ voicemail="on" if voicemail_enabled() else "off",
567
+ )
568
+ + f"\n\n## Your slice\n\nYou are writing only: {mine.named()}"
569
+ ),
570
+ servers={
571
+ SCENARIO_SERVER: ToolServer(
572
+ name=server.name,
573
+ version=server.version,
574
+ tools=[spec for spec in server.tools if spec.name != "save_scenarios"],
575
+ )
576
+ },
577
+ cwd=str(destination.parent if destination.parent.exists() else Path.cwd()),
578
+ max_turns=turns_for(mine.count),
579
+ model=chosen_model(),
580
+ ask=ask,
581
+ thinking=True,
582
+ idle_timeout_seconds=QUIET_WHILE_WRITING_SECONDS,
583
+ )
584
+ # Each writer drives its own model session, so several of them together are a request rate. When
585
+ # the provider refuses one, the slice is not wrong and its brief is still worth writing, so wait
586
+ # for the quota to recover and run it again. `kept` belongs to this slice's own tool server, so a
587
+ # second attempt adds to what the first proved rather than starting over.
588
+ async def attempt_slice() -> Any:
589
+ stage = Stage(sliced, name=f"{SKILL}:{mine.named()[:40]}")
590
+ async with stage:
591
+ return await stage.say(
592
+ brief_for(contract, mine, siblings, callers_for(index, mine.count)),
593
+ on_event=watch,
594
+ )
595
+
596
+ try:
597
+ # A refusal for rate or quota is waited out; the slice keeps what it already proved, so a
598
+ # second attempt adds to it. Stop early if the count is already met.
599
+ await survive_refusal(
600
+ attempt_slice,
601
+ what=f"slice {mine.named()}",
602
+ on_event=on_event,
603
+ enough=lambda: len(kept) >= mine.count,
604
+ )
605
+ except Exception as broke: # noqa: BLE001 - one slice failing must not lose the others
606
+ logger.warning("slice %s failed after %s: %s", mine.named(), len(kept), broke)
607
+ if on_event:
608
+ on_event({"type": "slice_failed", "slice": mine.named(), "why": str(broke)[:300]})
609
+ return list(kept)
610
+ logger.info("slice %s finished with %s of %s", mine.named(), len(kept), mine.count)
611
+ return list(kept)
612
+
613
+
614
+ def merged(written: list[list[Scenario]]) -> list[Scenario]:
615
+ """One suite out of several writers, with folder-name collisions renamed rather than dropped.
616
+
617
+ Asking for twenty scenarios has to return twenty. Two scenarios may legitimately share a use
618
+ case and a branch and still test different things, so sharing them is not a reason to discard
619
+ one; an earlier version dropped those and quietly returned eighteen.
620
+
621
+ The one collision that cannot be tolerated is the folder name, because the folder is where a
622
+ scenario lives on disk and the loser would overwrite the winner. Those are given a numbered
623
+ suffix instead of being thrown away, so nothing generated is ever lost.
624
+ """
625
+ suite: list[Scenario] = []
626
+ taken: set[str] = set()
627
+ for batch in written:
628
+ for one in batch:
629
+ if one.name in taken:
630
+ stem, suffix = one.name, 2
631
+ while f"{stem}-{suffix}" in taken:
632
+ suffix += 1
633
+ one = one.model_copy(update={"name": f"{stem}-{suffix}", "scenario_key": ""})
634
+ logger.info("renamed a duplicate folder name to %s", one.name)
635
+ taken.add(one.name)
636
+ suite.append(one)
637
+ return suite
638
+
639
+
640
+ def _suite_summary(suite: list[Scenario]) -> str:
641
+ """The whole suite as a reviewer needs to see it: what each row claims to test."""
642
+ return "\n".join(
643
+ f" {one.name} | use case: {one.use_case} | branch: {one.branch} | passes when: {one.tests}"
644
+ for one in suite
645
+ )
646
+
647
+
648
+ async def gaps_in(
649
+ contract: AgentContract,
650
+ suite: list[Scenario],
651
+ *,
652
+ destination: Path,
653
+ wanted: int,
654
+ ask: Callable[..., Any] | None = None,
655
+ ) -> list[Slice]:
656
+ """What the finished suite is missing, as slices that would fill it.
657
+
658
+ Nobody looks at a suite written in parallel. Each writer sees its own slice and the merge
659
+ only removes collisions, so a use case that came back one short, or an obvious branch that
660
+ every writer assumed somebody else had, survives to the end and nobody notices. This is the
661
+ one pass that reads the suite as a whole.
662
+ """
663
+ if not suite:
664
+ return []
665
+ found: list[Slice] = []
666
+
667
+ @tool(
668
+ "submit_gaps",
669
+ "The gaps worth filling in this suite, as the slices that would fill them. Return "
670
+ "nothing when the suite covers what it should: a suite that is finished is a real "
671
+ "answer, and inventing work to report is worse than saying so.",
672
+ schema(
673
+ {
674
+ "gaps": {
675
+ "type": "array",
676
+ "description": "One entry per gap. Empty when the suite is covering what "
677
+ "it should.",
678
+ "items": {
679
+ "type": "object",
680
+ "properties": {
681
+ "use_case": {"type": "string"},
682
+ "angle": {
683
+ "type": "string",
684
+ "description": "The scenario that is missing, in one line.",
685
+ },
686
+ "why": {"type": "string"},
687
+ },
688
+ "required": ["use_case", "angle"],
689
+ },
690
+ }
691
+ },
692
+ ["gaps"],
693
+ ),
694
+ )
695
+ async def submit_gaps(args: dict[str, Any]) -> dict[str, Any]:
696
+ for one in args.get("gaps") or []:
697
+ if not isinstance(one, dict):
698
+ continue
699
+ case = str(one.get("use_case") or "").strip()
700
+ if case:
701
+ found.append(
702
+ Slice(
703
+ use_case=case,
704
+ angle=str(one.get("angle") or "").strip(),
705
+ count=1,
706
+ why=str(one.get("why") or "").strip(),
707
+ )
708
+ )
709
+ return {
710
+ "content": [
711
+ {"type": "text", "text": f"{len(found)} gap(s) recorded. Nothing else to do."}
712
+ ]
713
+ }
714
+
715
+ server = tool_server(name=REVIEW_SERVER, version="0.1.0", tools=[submit_gaps])
716
+ review = SessionSpec(
717
+ system_prompt=(
718
+ "You are reviewing a suite of tests somebody else wrote for an AI agent, in "
719
+ "parallel, each writer blind to the others. Your only job is to say what is "
720
+ "missing.\n\n"
721
+ "Look for: a use case of this agent that nothing covers; a use case covered only "
722
+ "on its ordinary path, where the branch that cannot be completed or the rule under "
723
+ "pressure is the interesting one; two rows that are the same test under different "
724
+ "names, leaving the branch one of them claimed uncovered.\n\n"
725
+ "Judge coverage of the agent, not of the plan. Do not ask for more of what is "
726
+ "already well covered, and do not report a gap you cannot name a scenario for. "
727
+ "A suite of the right size that covers what matters is finished, and saying so is "
728
+ f"the useful answer.\n\n## This agent\n\n{contract.brief()}"
729
+ ),
730
+ servers={REVIEW_SERVER: server},
731
+ cwd=str(destination.parent if destination.parent.exists() else Path.cwd()),
732
+ max_turns=8,
733
+ model=chosen_model(),
734
+ ask=ask,
735
+ )
736
+ stage = Stage(review, name=f"{SKILL}:review")
737
+ try:
738
+ async with stage:
739
+ # The suite-level checks are reported at save and enforced nowhere, deliberately:
740
+ # refusing a save leaves proved work in memory only. This is the one place that can act
741
+ # on them, because it is the only pass that reads the whole suite and can commission
742
+ # replacements. Without this they are a message nobody reads.
743
+ skew = suite_diversity_problems(suite)
744
+ already_wrong = (
745
+ "Reading the suite as a whole, these are already wrong with it, and a gap that "
746
+ "fixes one of them is worth more than a new use case:\n"
747
+ + "\n".join(f"- {problem}" for problem in skew)
748
+ + "\n\n"
749
+ if skew
750
+ else ""
751
+ )
752
+ asked = (
753
+ f"This suite has {len(suite)} scenarios against a target of {wanted}:\n\n"
754
+ f"{_suite_summary(suite)}\n\n"
755
+ + already_wrong
756
+ + "Say what it is missing, then submit_gaps. Submit an empty list if it is "
757
+ "covering what it should."
758
+ )
759
+ await survive_refusal(
760
+ lambda: stage.say(asked), what="the suite review", on_event=on_event
761
+ )
762
+ except Exception: # noqa: BLE001 - a review that fails leaves the suite as written
763
+ return []
764
+ return found
765
+
766
+
767
+ async def write_in_parallel(
768
+ contract: AgentContract,
769
+ *,
770
+ out: Path | None = None,
771
+ wanted: int = 10,
772
+ use_cases: list[str] | None = None,
773
+ slices: list[dict] | None = None,
774
+ at_once: int = AT_ONCE,
775
+ rounds: int = TOP_UP_ROUNDS,
776
+ on_event: Callable[..., Any] | None = None,
777
+ ask: Callable[..., Any] | None = None,
778
+ ) -> list[Scenario]:
779
+ """Write a suite with one session per slice, review it, fill what it missed, and save once.
780
+
781
+ Sequentially, a suite costs roughly three turns a scenario against one budget, which is why
782
+ asking for forty stopped around twenty-five. Here the work is split into slices that run at
783
+ the same time, so the wall clock is the slowest slice rather than the sum of all of them.
784
+
785
+ Saving stays here, once, for a reason: ``save_scenarios`` regenerates the index and deletes
786
+ any folder it does not know about, so letting the writers save would have each of them
787
+ remove the others' work.
788
+ """
789
+ destination = out or artifact_dir(contract.agent)
790
+ cases = [case for case in (use_cases or contract.real_use_cases) if case.strip()]
791
+ if not cases and not slices:
792
+ # Nothing to partition on. One writer, the ordinary path, rather than no scenarios.
793
+ return await write(contract, out=destination, wanted=wanted, on_event=on_event, ask=ask)
794
+
795
+ at_once = max(1, min(at_once or AT_ONCE, MOST_AT_ONCE))
796
+ allocation = planned(wanted, cases, slices)
797
+ logger.info(
798
+ "writing %s scenarios across %s slices, %s at a time: %s",
799
+ wanted,
800
+ len(allocation),
801
+ at_once,
802
+ ", ".join(f"{one.named()} x{one.count}" for one in allocation),
803
+ )
804
+ if on_event:
805
+ on_event(
806
+ {
807
+ "type": "planned",
808
+ "slices": [(one.named(), one.count) for one in allocation],
809
+ "at_once": at_once,
810
+ }
811
+ )
812
+
813
+ limit = asyncio.Semaphore(at_once)
814
+
815
+ async def guarded(mine: Slice, siblings: list[Slice], index: int) -> list[Scenario]:
816
+ async with limit:
817
+ return await _write_slice(
818
+ contract,
819
+ mine,
820
+ siblings,
821
+ index=index,
822
+ destination=destination,
823
+ on_event=on_event,
824
+ ask=ask,
825
+ )
826
+
827
+ written = await asyncio.gather(
828
+ *(guarded(one, allocation, index) for index, one in enumerate(allocation)),
829
+ return_exceptions=False,
830
+ )
831
+ proved = merged([load_scenarios(destination), *written])
832
+ # What a writer proved but never handed back, because its session died after proving it. Matched
833
+ # by name so a scenario already in hand is not added twice under a numbered name.
834
+ recovered = [
835
+ one for one in journalled(destination) if one.name not in {x.name for x in proved}
836
+ ]
837
+ if recovered:
838
+ logger.warning(
839
+ "recovered %s scenarios from the journal that no writer returned: %s",
840
+ len(recovered),
841
+ ", ".join(one.name for one in recovered),
842
+ )
843
+ if on_event:
844
+ on_event({"type": "recovered", "kept": len(recovered)})
845
+ suite = merged([proved, recovered])
846
+
847
+ # Read the whole thing and fill what nobody covered. Bounded, because a reviewer asked
848
+ # twice will always find something smaller to say.
849
+ for _ in range(max(0, rounds)):
850
+ if len(suite) >= wanted:
851
+ break
852
+ missing = await gaps_in(
853
+ contract, suite, destination=destination, wanted=wanted, ask=ask
854
+ )
855
+ missing = missing[: max(0, wanted - len(suite))]
856
+ if not missing:
857
+ break
858
+ if on_event:
859
+ on_event({"type": "topping_up", "slices": [one.named() for one in missing]})
860
+ logger.info(
861
+ "topping up %s of %s with %s more slices: %s",
862
+ len(suite),
863
+ wanted,
864
+ len(missing),
865
+ ", ".join(f"{one.named()} x{one.count}" for one in missing),
866
+ )
867
+ more = await asyncio.gather(
868
+ *(
869
+ guarded(one, missing, len(allocation) + index)
870
+ for index, one in enumerate(missing)
871
+ ),
872
+ return_exceptions=False,
873
+ )
874
+ before = len(suite)
875
+ suite = merged([suite, *more])
876
+ allocation = [*allocation, *missing]
877
+ if len(suite) == before:
878
+ break
879
+
880
+ write_scenarios(suite, destination, load_catalogue(destination))
881
+ logger.info("suite saved: %s of %s asked for", len(suite), wanted)
882
+ if on_event:
883
+ on_event({"type": "saved", "kept": len(suite), "asked": wanted})
884
+ return load(destination)
885
+
886
+
887
+ async def write(
888
+ contract: AgentContract,
889
+ *,
890
+ out: Path | None = None,
891
+ wanted: int = 10,
892
+ follow_ups: list[str] | None = None,
893
+ on_event: Callable[..., Any] | None = None,
894
+ ask: Callable[..., Any] | None = None,
895
+ max_turns: int = 0,
896
+ ) -> list[Scenario]:
897
+ """Run the stage start to finish. Returns whatever scenarios were saved."""
898
+ stage, destination = open_stage(
899
+ contract, out=out, wanted=wanted, ask=ask, max_turns=max_turns
900
+ )
901
+ async with stage:
902
+ # The planning turn is the expensive one to lose: a refusal here costs the whole suite, not
903
+ # one slice, so it waits the same way a writer does.
904
+ await survive_refusal(
905
+ lambda: stage.say(opening(contract, wanted), on_event=on_event),
906
+ what="the opening turn",
907
+ on_event=on_event,
908
+ )
909
+ for follow_up in follow_ups or []:
910
+ await survive_refusal(
911
+ lambda message=follow_up: stage.say(message, on_event=on_event),
912
+ what="a follow-up turn",
913
+ on_event=on_event,
914
+ )
915
+ return load(destination)