agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,2218 @@
1
+ """World pool + scenario loop — `hosted-execution-seams.md` v1.12 §4/§5, `world-handle-interface.md`
2
+ v3.4, `outbound-channels.md` v1.3.
3
+
4
+ Owns: leasing/releasing the W worlds `process_runtime.ProcessRuntimeProvider.provision()` hands
5
+ back, resetting a world to pristine before each scenario (spine §4.2), running one scenario's
6
+ `setup`/`ready`/checks against the world handle (the return-convention + errored-receipt table in
7
+ `world-handle-interface.md`), the fixed one-retry-on-a-fresh-world rule (spine §5 step 4), and
8
+ synthesizing a complete receipt ledger (one per scenario, `skipped` for anything never attempted).
9
+
10
+ Decoupling, deliberate:
11
+ - `process_runtime.py` is the real, settled provisioner — imported directly (`EnvironmentRuntime`,
12
+ `RuntimeState`). `WorldProvisioner` below is a structural `Protocol` matching
13
+ `ProcessRuntimeProvider`'s actual async shape so tests can inject a fake without touching a real
14
+ filesystem/subprocess tree.
15
+ - `OutboundPort` is this module's own minimal sink for the events/receipts it produces, typed
16
+ against `outbound-channels.md`'s closed vocabulary; whoever wires the real client adapts to it.
17
+ `outbound.py` is now committed and quiescent, so this module imports exactly three of its
18
+ exception types — `HostedFencedError`/`HostedChannelFailedError`/`HostedAttemptSupersededError`,
19
+ the full `ChannelState` "stop emitting" latch for one attempt — to recognize the one outbound
20
+ failure class that is NOT best-effort (a 401/403 fence, an exhausted channel, or a superseded
21
+ attempt must stop the run, not be logged and forgotten); nothing else from that module is
22
+ imported here.
23
+ - The Scenario Generation Contract (in review) is not available here either, so `Scenario`
24
+ is this module's own minimal Protocol for what the loop needs: a key/id pair, `setup`/`ready`,
25
+ and named sub-goal checks. Same for the simulated "call" itself (a different track's seam) —
26
+ `CallRunner` is injected.
27
+ - Secrets and the cancel signal are entrypoint-owned (P10); `cancel_requested` is an injected
28
+ zero-argument callable.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import asyncio
34
+ import inspect
35
+ import logging
36
+ import random
37
+ import re
38
+ import threading
39
+ from concurrent.futures import ThreadPoolExecutor
40
+ from dataclasses import dataclass
41
+ from pathlib import Path
42
+ from typing import Any, Awaitable, Callable, Protocol, Sequence
43
+
44
+ from . import observability
45
+ from .job import FailureDomain, HarnessStage
46
+ from .judge import judge as _judge
47
+ from .outbound import (
48
+ HostedAttemptSupersededError,
49
+ HostedChannelFailedError,
50
+ HostedFencedError,
51
+ )
52
+ from .process_runtime import (
53
+ SECTION_2F_DOMAIN,
54
+ EnvironmentRuntime,
55
+ ProcessRuntimeError,
56
+ RuntimeState,
57
+ )
58
+
59
+ from .world.errors import (
60
+ WorldError,
61
+ WorldQueryRejected,
62
+ WorldReadOnly,
63
+ WorldReservedName,
64
+ WorldStateTooLarge,
65
+ WorldUnavailable,
66
+ WorldUsageError,
67
+ )
68
+ from .world.runtime import Call
69
+
70
+ logger = logging.getLogger(__name__)
71
+
72
+ # --- the World handle (world-handle-interface.md v3.4) --------------------------------------
73
+ #
74
+ # The frozen contract's code block gives six verbs plus `world_index`/`rng`. `read_only()` is not
75
+ # in that block, but the contract still requires `ready`/`check` to receive a handle whose writes
76
+ # raise `WorldReadOnly` (own section, "Read-only handles") without saying how a caller gets one —
77
+ # the shipped `HostedWorld.read_only()` (`world/handle.py`) already names this exact operation, so
78
+ # mirroring it here is the reversible choice: a real `HostedWorld` satisfies this Protocol as-is.
79
+ #
80
+ # m8: `World.read_only()` used to be typed `-> "World"`, but the real `ReadOnlyWorld` it returns
81
+ # has no `read_only()` of its own (mirroring `world/handle.py`'s own `ReadOnlyWorld`, which is
82
+ # deliberately not re-enterable) — so it fails a structural check against `World` itself.
83
+ # `ReadOnlyWorld` below names the narrower surface `ready`/`check` actually receive.
84
+
85
+
86
+ class ReadOnlyWorld(Protocol):
87
+ world_index: int
88
+ rng: random.Random
89
+
90
+ def state(self, table: str | None = None) -> dict[str, list[dict[str, Any]]]: ...
91
+
92
+ def put(
93
+ self, collection: str, record: dict[str, Any], *, key: str = ""
94
+ ) -> dict[str, Any]: ...
95
+
96
+ def change(
97
+ self, collection: str, key: str, changes: dict[str, Any], *, by: str = ""
98
+ ) -> int: ...
99
+
100
+ def drop(self, collection: str, key: str = "", *, by: str = "") -> int: ...
101
+
102
+ def call(self, name: str, arguments: dict[str, Any] | None = None) -> Call: ...
103
+
104
+ def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]: ...
105
+
106
+
107
+ class World(Protocol):
108
+ world_index: int
109
+ rng: random.Random
110
+
111
+ def state(self, table: str | None = None) -> dict[str, list[dict[str, Any]]]: ...
112
+
113
+ def put(
114
+ self, collection: str, record: dict[str, Any], *, key: str = ""
115
+ ) -> dict[str, Any]: ...
116
+
117
+ def change(
118
+ self, collection: str, key: str, changes: dict[str, Any], *, by: str = ""
119
+ ) -> int: ...
120
+
121
+ def drop(self, collection: str, key: str = "", *, by: str = "") -> int: ...
122
+
123
+ def call(self, name: str, arguments: dict[str, Any] | None = None) -> Call: ...
124
+
125
+ def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]: ...
126
+
127
+ def read_only(self) -> ReadOnlyWorld: ...
128
+
129
+
130
+ class WorldFactory(Protocol):
131
+ """Builds the `World` handle for one already-reset `EnvironmentRuntime`.
132
+
133
+ Deliberately not this module's job: `HostedWorld` needs a `PostgresStore` (parsed from the
134
+ runtime's `database` endpoint) plus the baseline row counts the provisioner measured at
135
+ freeze time — both live behind `ProcessRuntimeProvider`'s private state, which §4's
136
+ `RuntimeProvider` Protocol never exposes. Injected instead of guessed.
137
+ """
138
+
139
+ async def create(
140
+ self, runtime: EnvironmentRuntime, *, rng: random.Random
141
+ ) -> World: ...
142
+
143
+
144
+ # --- the provisioner surface this module actually drives -------------------------------------
145
+ #
146
+ # Matches `ProcessRuntimeProvider`'s real async shape (process_runtime.py) structurally, not the
147
+ # older single-runtime `runtime.RuntimeProvider`. `bundle` stays `Any` — this module never reads a
148
+ # bundle field itself, only threads it back into `provision()` reconcile calls, so it does not
149
+ # need `EnvironmentBundleV2`'s own in-flux-adjacent type.
150
+ #
151
+ # M1 (spine v1.12 §4): `bundle_dir` is a required keyword — §2c seed/migration paths resolve
152
+ # against the verified bundle root, never against `source`. `require_declared_user` is dropped
153
+ # entirely: it is not in §4's signature, and the real provider now defaults it `True` on its own
154
+ # (the local lane opts out at provider construction, not per call).
155
+ # M2 (spine §4 point 3): `healthy` — declared readiness probes, not "process is running" — is a
156
+ # port method, not an optionally-injected callable.
157
+
158
+
159
+ class WorldProvisioner(Protocol):
160
+ async def provision(
161
+ self,
162
+ bundle: Any,
163
+ *,
164
+ source: Path,
165
+ bundle_dir: Path,
166
+ work_directory: Path,
167
+ contract: Any | None = None,
168
+ instances: int = 1,
169
+ ) -> list[EnvironmentRuntime]: ...
170
+
171
+ async def reset(
172
+ self, runtime: EnvironmentRuntime, *, work_directory: Path
173
+ ) -> None: ...
174
+
175
+ async def healthy(
176
+ self, runtime: EnvironmentRuntime, *, work_directory: Path
177
+ ) -> bool: ...
178
+
179
+ async def close(self, *, work_directory: Path) -> None: ...
180
+
181
+
182
+ # --- scenarios (this module's own minimal surface; that contract is not wired yet) -------
183
+
184
+
185
+ class SubGoal(Protocol):
186
+ name: str
187
+ judged: (
188
+ str # `sub_goals[].judged` per outbound-channels.md: boolean is `judged != ""`.
189
+ )
190
+
191
+ def check(self, world: ReadOnlyWorld, calls: Sequence[Call]) -> object: ...
192
+
193
+
194
+ JudgeFn = Callable[..., Awaitable[tuple[bool | None, str]]]
195
+
196
+
197
+ class Scenario(Protocol):
198
+ scenario_key: str
199
+ scenario_id: (
200
+ str # platform id from pre-allocation (outbound-channels.md Channel 2 "Join").
201
+ )
202
+ sub_goals: Sequence[SubGoal]
203
+ # True only when the reference solution contains an environment/tool action. Pure
204
+ # conversation scenarios (for example, refusing an unsafe request) legitimately produce no
205
+ # world calls and must still reach their judged checks.
206
+ requires_tool_evidence: bool
207
+
208
+ def setup(self, world: World) -> object: ...
209
+
210
+ def ready(self, world: ReadOnlyWorld) -> object: ...
211
+
212
+
213
+ # --- the simulated call (a different track's seam; injected, never built here) ---------------
214
+
215
+
216
+ @dataclass(frozen=True)
217
+ class CallOutcome:
218
+ calls: tuple[Call, ...]
219
+ turns: int
220
+ started_at: str | None
221
+ ended_at: str | None
222
+ duration_ms: int
223
+ transcript_artifact: str | None = None
224
+ recording_artifacts: tuple[str, ...] = ()
225
+ stop_reason: str | None = None
226
+ # The artifact above is an id the sandbox cannot read back.
227
+ messages: tuple[Any, ...] = ()
228
+
229
+
230
+ class CallAborted(RuntimeError):
231
+ """The call step started but did not finish. `partial`, when known, carries whatever timing
232
+ the call runner already measured — the receipt's `call` field must not be null once the call
233
+ has genuinely started (outbound-channels.md Channel 2, "errored receipt body")."""
234
+
235
+ def __init__(
236
+ self,
237
+ message: str,
238
+ *,
239
+ partial: CallOutcome | None = None,
240
+ code: str = "call_failed",
241
+ ) -> None:
242
+ super().__init__(message)
243
+ self.partial = partial
244
+ self.code = code
245
+
246
+
247
+ class CallRunner(Protocol):
248
+ async def run(
249
+ self,
250
+ scenario: Scenario,
251
+ runtime: EnvironmentRuntime,
252
+ *,
253
+ world: World | None = None,
254
+ ) -> CallOutcome: ...
255
+
256
+
257
+ async def _run_call(
258
+ runner: CallRunner,
259
+ scenario: Scenario,
260
+ runtime: EnvironmentRuntime,
261
+ world: World,
262
+ ) -> CallOutcome:
263
+ """Pass the world to text runners while preserving older two-argument integrations.
264
+
265
+ Voice runners do not execute response-carried tools themselves. Hosted HTTP chat runners do,
266
+ and must execute them against the exact leased world that setup/checks observe. The signature
267
+ probe keeps the settled injected-runner seam source compatible for downstream callers while
268
+ allowing that missing context to cross the boundary.
269
+ """
270
+ run = runner.run
271
+ parameters = inspect.signature(run).parameters.values()
272
+ accepts_world = any(
273
+ parameter.name == "world" or parameter.kind is inspect.Parameter.VAR_KEYWORD
274
+ for parameter in parameters
275
+ )
276
+ if accepts_world:
277
+ return await run(scenario, runtime, world=world)
278
+ return await run(scenario, runtime)
279
+
280
+
281
+ # --- receipts (outbound-channels.md Channel 2; envelope fields — job_id/attempt_id/digest/etc —
282
+ # are the emitter's concern, not reproduced here) -----------------------------------------------
283
+
284
+
285
+ @dataclass(frozen=True)
286
+ class SubGoalResult:
287
+ name: str
288
+ held: bool | None
289
+ reason: str | None
290
+ judged: bool
291
+
292
+
293
+ @dataclass(frozen=True)
294
+ class Evaluation:
295
+ name: str
296
+ kind: str # "metric" | "checkpoint"
297
+ reason: str
298
+ score: float | None = None
299
+ passed: bool | None = None
300
+
301
+
302
+ @dataclass(frozen=True)
303
+ class CallSummary:
304
+ started_at: str | None
305
+ ended_at: str | None
306
+ duration_ms: int
307
+ turns: int
308
+ transcript_artifact: str | None = None
309
+ recording_artifacts: tuple[str, ...] = ()
310
+ stop_reason: str | None = None
311
+
312
+
313
+ @dataclass(frozen=True)
314
+ class ReceiptFailure:
315
+ domain: str
316
+ stage: str
317
+ code: str
318
+ message: str
319
+
320
+
321
+ @dataclass(frozen=True)
322
+ class ResultReceipt:
323
+ scenario_key: str
324
+ scenario_id: str
325
+ scenario_attempt: int
326
+ world_index: int | None
327
+ status: str # "passed" | "failed" | "errored" | "skipped"
328
+ sub_goals: tuple[SubGoalResult, ...]
329
+ evaluations: tuple[Evaluation, ...]
330
+ call: CallSummary | None
331
+ failure: ReceiptFailure | None
332
+
333
+
334
+ # --- outbound (this module's own minimal sink; see the module docstring's decoupling note) ----
335
+
336
+
337
+ class OutboundPort(Protocol):
338
+ async def scenario_started(
339
+ self, *, scenario_key: str, world_index: int, scenario_attempt: int
340
+ ) -> None: ...
341
+
342
+ async def scenario_retried(
343
+ self, *, scenario_key: str, from_world: int, to_world: int
344
+ ) -> None: ...
345
+
346
+ async def world_unhealthy(self, *, world_index: int, cause: str) -> None: ...
347
+
348
+ async def log(self, *, level: str, message: str) -> None: ...
349
+
350
+ async def receipt(self, receipt: ResultReceipt) -> None: ...
351
+
352
+
353
+ # --- failure-code -> FailureDomain, and which codes retry once on a fresh world ----------------
354
+ #
355
+ # `world_unavailable` is domain `environment` per world-handle-interface.md's own errored-receipt
356
+ # table (it overrides the table's default "domain: simulator"). `evidence_missing` is domain
357
+ # simulator but is explicitly carved out as retryable in that same document ("gets the same single
358
+ # retry-on-another-world as a world failure"). `call_failed` (v3.3) and `driver_crashed` (v3.4) are
359
+ # both rows in that same closed table now: `call_failed` is domain infrastructure, retried once
360
+ # like a world failure; `driver_crashed` is domain simulator, not retried — the scheduler's own
361
+ # machinery failing while driving a scenario, distinct from any agent/check/call outcome.
362
+ # `world_pool_exhausted` is NOT a per-scenario receipt code — it is `HostedScheduler.run()`'s own
363
+ # job-abort signal for spine v1.12 §5.4's closed job-level failure vocabulary for stage `running`
364
+ # (domain infrastructure).
365
+
366
+ _CODE_DOMAIN: dict[str, FailureDomain] = {
367
+ "setup_crashed": FailureDomain.SIMULATOR,
368
+ "setup_timeout": FailureDomain.SIMULATOR,
369
+ "ready_timeout": FailureDomain.SIMULATOR,
370
+ "check_timeout": FailureDomain.SIMULATOR,
371
+ "ready_not_ready": FailureDomain.SIMULATOR,
372
+ "ready_broken": FailureDomain.SIMULATOR,
373
+ "check_broken": FailureDomain.SIMULATOR,
374
+ "judge_undecided": FailureDomain.SIMULATOR,
375
+ "evidence_missing": FailureDomain.SIMULATOR,
376
+ "world_usage": FailureDomain.SIMULATOR,
377
+ "world_unavailable": FailureDomain.ENVIRONMENT,
378
+ "state_too_large": FailureDomain.SIMULATOR,
379
+ "call_failed": FailureDomain.INFRASTRUCTURE,
380
+ "target_agent_stalled": FailureDomain.AGENT,
381
+ "target_agent_tool_failed": FailureDomain.AGENT,
382
+ "simulator_stalled": FailureDomain.SIMULATOR,
383
+ "driver_crashed": FailureDomain.SIMULATOR,
384
+ "world_pool_exhausted": FailureDomain.INFRASTRUCTURE,
385
+ }
386
+ _RETRYABLE_CODES = frozenset(
387
+ {"evidence_missing", "target_agent_stalled", "simulator_stalled"}
388
+ )
389
+
390
+ # hosted-execution-seams.md v1.13 §5.4/§2f: the closed provisioner build/run failure-code table --
391
+ # these used to be discarded at the reset()/provision() seam (caught as a bare `Exception`, only
392
+ # `str()` surviving into `world_unhealthy.cause`), so a deterministic `environment`/`agent` fault
393
+ # (never retried) was re-reported as `world_pool_exhausted`/infrastructure and burned every
394
+ # whole-job retry on a failure that repeats identically. v1.15: the PRODUCER now resolves
395
+ # `spawn_failed`'s managed-vs-source split at the raise site (`ProcessRuntimeError.domain`) --
396
+ # this module reads that carried domain first; `SECTION_2F_DOMAIN` (imported from
397
+ # `process_runtime.py`, the codes' own home) is consulted only for the rare error that reaches
398
+ # here with no carried domain, and that fallback is logged so a silent re-guess never hides again.
399
+
400
+
401
+ def _resolve_2f_domain(
402
+ code: str | None,
403
+ domain: FailureDomain | None,
404
+ ) -> tuple[str, FailureDomain] | None:
405
+ """v1.15 §2f: pair a code with its resolved domain. `domain` should be the value CARRIED by a
406
+ typed provisioner error (`ProcessRuntimeError.domain`); `None` here falls back to the closed
407
+ code->domain map (and logs it) rather than silently re-guessing. Returns `None` outright for
408
+ no code, or a code outside the §2f table (`internal_*` etc.) — callers already treat that the
409
+ same as "not a §2f error."
410
+ """
411
+ if code is None or code not in SECTION_2F_DOMAIN:
412
+ return None
413
+ if domain is not None:
414
+ return code, domain
415
+ logger.warning(
416
+ "process_runtime error %r crossed the §4 seam with no carried domain; using the §2f "
417
+ "fallback map (%s)",
418
+ code,
419
+ SECTION_2F_DOMAIN[code].value,
420
+ )
421
+ return code, SECTION_2F_DOMAIN[code]
422
+
423
+
424
+ # v1.13: only these two domains are "never retried" -- a uniform §2f code across every unhealthy
425
+ # world in one of them surfaces as that code+domain; anything else (mixed codes, or any
426
+ # infrastructure-domain fault) stays `world_pool_exhausted` exactly as before.
427
+ _SECTION_2F_NEVER_RETRIED = frozenset({FailureDomain.ENVIRONMENT, FailureDomain.AGENT})
428
+
429
+ # The one `OutboundPort` failure class that is NOT best-effort -- outbound-channels.md:
430
+ # 401/403 -> "stop emitting, exit code 3 ... never an infra retry," and the same latch also covers
431
+ # a 404-exhausted channel and a 409 attempt-supersession (outbound.py's `ChannelState`: "a fence in
432
+ # substance"). Letting `_emit`/`_log`/`mark_unhealthy` swallow any of these the same way they
433
+ # swallow a transport hiccup ran a superseded attempt's entire scenario set after the platform had
434
+ # already fenced or superseded it.
435
+ _FATAL_OUTBOUND: tuple[type[Exception], ...] = (
436
+ HostedFencedError,
437
+ HostedChannelFailedError,
438
+ HostedAttemptSupersededError,
439
+ )
440
+
441
+ # M13: an exception/overrun outcome leaves the world half-applied — world-handle-interface.md's
442
+ # return-conventions rule is "the world is discarded and re-provisioned (a half-applied world is
443
+ # never reused)." `ready_not_ready` is deliberately excluded: a precondition failing on the shared
444
+ # sealed baseline is a clean verdict, not an exception, so the world itself is still fine.
445
+ _DISCARD_ON_ERROR_CODES = frozenset(
446
+ {
447
+ "setup_crashed",
448
+ "setup_timeout",
449
+ "ready_timeout",
450
+ "check_timeout",
451
+ "ready_broken",
452
+ "check_broken",
453
+ "world_usage",
454
+ "state_too_large",
455
+ }
456
+ )
457
+
458
+ SETUP_TIMEOUT_SECONDS = 60.0
459
+ READY_TIMEOUT_SECONDS = 15.0
460
+ CHECK_TIMEOUT_SECONDS = 60.0
461
+
462
+ _MESSAGE_LIMIT = 2000 # matches the Call.result/error truncation convention (world-handle-interface.md).
463
+ _CAUSE_LIMIT = (
464
+ 200 # outbound-channels.md Channel 1: `world_unhealthy.cause` free text <=200.
465
+ )
466
+ _USERINFO_PATTERN = re.compile(r"://[^@/]+@")
467
+
468
+
469
+ def _is_retryable(code: str) -> bool:
470
+ return _CODE_DOMAIN[code] in (
471
+ FailureDomain.ENVIRONMENT,
472
+ FailureDomain.INFRASTRUCTURE,
473
+ ) or (code in _RETRYABLE_CODES)
474
+
475
+
476
+ def _truncate(text: str, limit: int = _MESSAGE_LIMIT) -> str:
477
+ if len(text) <= limit:
478
+ return text
479
+ return text[: limit - len("…[truncated]")] + "…[truncated]"
480
+
481
+
482
+ def _sanitize_cause(message: str) -> str:
483
+ # M7: `cause` is capped at 200 chars and must never carry endpoint credentials — postgres
484
+ # error strings routinely embed the DSN (`postgresql://user:pw@host/db`).
485
+ return _truncate(_USERINFO_PATTERN.sub("://***@", message), _CAUSE_LIMIT)
486
+
487
+
488
+ def _failure(code: str, message: str) -> ReceiptFailure:
489
+ return ReceiptFailure(
490
+ domain=_CODE_DOMAIN[code].value,
491
+ stage=HarnessStage.RUNNING.value,
492
+ code=code,
493
+ message=_truncate(message),
494
+ )
495
+
496
+
497
+ # --- return-convention classification (world-handle-interface.md "Return conventions") --------
498
+
499
+
500
+ @dataclass(frozen=True)
501
+ class _Verdict:
502
+ held: bool
503
+ reason: str | None
504
+ broken: bool
505
+
506
+
507
+ def _classify_ready(value: object) -> _Verdict:
508
+ if value is None or value is True:
509
+ return _Verdict(True, None, False)
510
+ if isinstance(value, str):
511
+ if value.strip() == "":
512
+ return _Verdict(True, None, False)
513
+ return _Verdict(False, value, False)
514
+ # Bare False or any other value -> broken. checks.py's `run_world_check` treats a non-None,
515
+ # non-string ready() answer the same way; a scenario hitting this cannot be told apart from a
516
+ # buggy ready.py, which is why it is `ready_broken` rather than a clean not-ready verdict.
517
+ return _Verdict(False, None, True)
518
+
519
+
520
+ def _classify_check(value: object) -> _Verdict:
521
+ if value is None or value is True:
522
+ return _Verdict(True, None, False)
523
+ if isinstance(value, str):
524
+ if value.strip() == "":
525
+ return _Verdict(True, None, False)
526
+ return _Verdict(False, value, False)
527
+ if value is False:
528
+ # An agent result ("the agent did something wrong"), not a broken check — matches
529
+ # checks.py's `Outcome(name, False, "False")`.
530
+ return _Verdict(False, "False", False)
531
+ return _Verdict(False, None, True)
532
+
533
+
534
+ def _sub_goal_reason(goal: SubGoal, verdict: _Verdict) -> str | None:
535
+ """What to show a reader for this sub-goal, on a pass as much as on a failure.
536
+
537
+ A check returns nothing when it holds, which left every passing sub-goal with an empty hover
538
+ and no way to tell a real pass from one nobody wrote a check for. The authored description of
539
+ what the sub-goal means is the honest thing to show there: it says what was verified without
540
+ claiming evidence the check never returned. A bare ``False`` is the other end of the same
541
+ problem -- the reason read literally "False" -- so it gets the description too.
542
+ """
543
+ what = str(getattr(goal, "what", "") or "").strip().rstrip(".")
544
+ if verdict.held:
545
+ return f"Held: {what}." if what else "Held. The check found nothing wrong."
546
+ if verdict.reason and verdict.reason.strip() and verdict.reason != "False":
547
+ return verdict.reason
548
+ return f"Did not hold: {what}." if what else None
549
+
550
+
551
+ # --- phase execution: budget + exception classification ---------------------------------------
552
+
553
+
554
+ class _PhaseTimeout(Exception):
555
+ def __init__(self, phase: str) -> None:
556
+ super().__init__(phase)
557
+ self.phase = phase
558
+
559
+
560
+ class _PhaseNeverStarted(Exception):
561
+ """R1: the phase's own worker thread had not even started running when its budget elapsed —
562
+ the dedicated executor was saturated, not the phase itself overrunning. Must not read as a
563
+ genuine timeout (which discards the world); the world did nothing wrong here."""
564
+
565
+ def __init__(self, phase: str) -> None:
566
+ super().__init__(phase)
567
+ self.phase = phase
568
+
569
+
570
+ class _PhaseWorldGone(Exception):
571
+ def __init__(self, phase: str, cause: BaseException) -> None:
572
+ super().__init__(f"{phase}: {cause}")
573
+ self.cause = cause
574
+
575
+
576
+ class _PhaseMisuse(Exception):
577
+ def __init__(self, phase: str, cause: BaseException) -> None:
578
+ super().__init__(f"{phase}: {cause}")
579
+ self.cause = cause
580
+
581
+
582
+ class _PhaseStateTooLarge(Exception):
583
+ def __init__(self, phase: str, cause: BaseException) -> None:
584
+ super().__init__(f"{phase}: {cause}")
585
+ self.cause = cause
586
+
587
+
588
+ class _PhaseCrashed(Exception):
589
+ def __init__(self, phase: str, cause: BaseException) -> None:
590
+ super().__init__(f"{phase}: {cause}")
591
+ self.cause = cause
592
+
593
+
594
+ async def _invoke(
595
+ fn: Callable[..., object],
596
+ *args: object,
597
+ timeout: float,
598
+ phase: str,
599
+ executor: ThreadPoolExecutor,
600
+ ) -> object:
601
+ # R1: a `threading.Event` set as the thread body's first statement — the only way to tell
602
+ # "the phase ran past its budget" (genuine overrun, world half-applied) apart from "the
603
+ # phase's thread was still queued behind others when the budget elapsed" (the scheduler's own
604
+ # executor was saturated; the world itself never touched anything).
605
+ started_flag = threading.Event()
606
+
607
+ def _run() -> object:
608
+ started_flag.set()
609
+ return fn(*args)
610
+
611
+ async def _call() -> object:
612
+ # B4: real scenario code (`setup`/`ready`/`check`) is synchronous, blocking psycopg calls
613
+ # — it must never run directly on the event loop, or the timeout below is purely
614
+ # decorative and every other world stalls with it. Dispatched to the scheduler's own
615
+ # dedicated executor (world-handle-interface.md: "one worker thread per world"; R1 —
616
+ # never the loop's default executor, which the provider's own `to_thread` calls also
617
+ # use). If the thread's own return value is itself awaitable (scenario code that is
618
+ # `async def`, reached indirectly through a sync wrapper), that coroutine is driven on
619
+ # the event loop afterward, where real suspension/cancellation actually works — this is
620
+ # the kept "awaitable" branch.
621
+ loop = asyncio.get_running_loop()
622
+ result = await loop.run_in_executor(executor, _run)
623
+ if inspect.isawaitable(result):
624
+ result = await result
625
+ return result
626
+
627
+ loop = asyncio.get_running_loop()
628
+ started = loop.time()
629
+ try:
630
+ return await asyncio.wait_for(_call(), timeout=timeout)
631
+ except asyncio.TimeoutError as exc:
632
+ # m4: `asyncio.TimeoutError is TimeoutError` on 3.11 — a psycopg statement timeout raised
633
+ # INSIDE `fn` looks identical to `wait_for`'s own deadline unless the elapsed time is
634
+ # actually checked. If the budget did not genuinely elapse, this was `fn`'s own timeout
635
+ # bubbling through — a broken phase, not a budget overrun.
636
+ if loop.time() - started < timeout:
637
+ raise _PhaseCrashed(phase, exc) from exc
638
+ if not started_flag.is_set():
639
+ raise _PhaseNeverStarted(phase) from exc
640
+ # B4: `wait_for`'s cancellation stops US from waiting on the thread, not the thread
641
+ # itself — psycopg in-flight cancellation is not wired here (P11 follow-up; recorded in
642
+ # the fixer report). The thread is abandoned, bounded by scenario count per the contract's
643
+ # own accepted tradeoff; its world is discarded rather than reused (M13).
644
+ raise _PhaseTimeout(phase) from exc
645
+ except WorldUnavailable as exc:
646
+ raise _PhaseWorldGone(phase, exc) from exc
647
+ except WorldStateTooLarge as exc:
648
+ raise _PhaseStateTooLarge(phase, exc) from exc
649
+ except (
650
+ WorldReadOnly,
651
+ WorldReservedName,
652
+ WorldQueryRejected,
653
+ WorldUsageError,
654
+ ) as exc:
655
+ raise _PhaseMisuse(phase, exc) from exc
656
+ except WorldError as exc:
657
+ # m5: catches any WorldError subclass not special-cased above (world/errors.py's own
658
+ # base, kept exactly for "route 'scenario code misused the handle' to one outcome without
659
+ # naming all six") — a future seventh subclass lands here instead of silently falling into
660
+ # the generic crash classification below.
661
+ raise _PhaseMisuse(phase, exc) from exc
662
+ except Exception as exc:
663
+ raise _PhaseCrashed(phase, exc) from exc
664
+
665
+
666
+ _CRASH_CODE_BY_PHASE = {
667
+ "setup": "setup_crashed",
668
+ "ready": "ready_broken",
669
+ "check": "check_broken",
670
+ }
671
+ _TIMEOUT_CODE_BY_PHASE = {
672
+ "setup": "setup_timeout",
673
+ "ready": "ready_timeout",
674
+ "check": "check_timeout",
675
+ }
676
+
677
+
678
+ @dataclass(frozen=True)
679
+ class _PhaseResult:
680
+ value: object
681
+ failure: ReceiptFailure | None
682
+
683
+
684
+ async def _run_phase(
685
+ fn: Callable[..., object],
686
+ *args: object,
687
+ timeout: float,
688
+ phase: str,
689
+ executor: ThreadPoolExecutor,
690
+ ) -> _PhaseResult:
691
+ try:
692
+ value = await _invoke(
693
+ fn, *args, timeout=timeout, phase=phase, executor=executor
694
+ )
695
+ return _PhaseResult(value, None)
696
+ except _PhaseNeverStarted:
697
+ # R1: not the phase's fault and not the world's — the scheduler's own thread pool
698
+ # couldn't service it in time. `driver_crashed` is not in `_DISCARD_ON_ERROR_CODES`, so
699
+ # this releases the world rather than discarding a perfectly healthy one.
700
+ return _PhaseResult(
701
+ None,
702
+ _failure(
703
+ "driver_crashed",
704
+ f"{phase} never started before its budget elapsed (thread pool saturated)",
705
+ ),
706
+ )
707
+ except _PhaseTimeout:
708
+ return _PhaseResult(
709
+ None,
710
+ _failure(_TIMEOUT_CODE_BY_PHASE[phase], f"{phase} exceeded its budget"),
711
+ )
712
+ except _PhaseWorldGone as exc:
713
+ return _PhaseResult(None, _failure("world_unavailable", str(exc.cause)))
714
+ except _PhaseMisuse as exc:
715
+ return _PhaseResult(None, _failure("world_usage", str(exc.cause)))
716
+ except _PhaseStateTooLarge as exc:
717
+ return _PhaseResult(None, _failure("state_too_large", str(exc.cause)))
718
+ except _PhaseCrashed as exc:
719
+ return _PhaseResult(
720
+ None,
721
+ _failure(
722
+ _CRASH_CODE_BY_PHASE[phase], f"{type(exc.cause).__name__}: {exc.cause}"
723
+ ),
724
+ )
725
+
726
+
727
+ # --- the world pool -----------------------------------------------------------------------------
728
+
729
+
730
+ class NoWorldsAvailable(RuntimeError):
731
+ """Every provisioned world is down and none is currently recoverable (spine v1.12 §5.4:
732
+ "if ready worlds reach 0 the job FAILS in stage running, domain infrastructure" — declared
733
+ only after in-flight re-provisioning completes without restoring a world), OR the pool has
734
+ been closed (R5: `reason="closed"`).
735
+
736
+ v1.13 §5.4: `code`/`domain` carry a uniform §2f never-retried code when every unhealthy
737
+ world's last re-provision attempt agreed on one — `None` (the default) means the caller falls
738
+ back to the generic `world_pool_exhausted`/infrastructure abort, exactly as before."""
739
+
740
+ def __init__(
741
+ self,
742
+ message: str,
743
+ *,
744
+ reason: str = "exhausted",
745
+ code: str | None = None,
746
+ domain: FailureDomain | None = None,
747
+ ) -> None:
748
+ super().__init__(message)
749
+ self.reason = reason
750
+ self.code = code
751
+ self.domain = domain
752
+
753
+
754
+ _RECONCILE_MAX_ATTEMPTS = 3
755
+ _RECONCILE_BACKOFF_SECONDS = (0.05, 0.1)
756
+ _LEASE_POLL_INTERVAL_SECONDS = 0.02
757
+
758
+
759
+ class WorldPool:
760
+ """Leases/releases the W worlds `provisioner.provision()` returns, resets one to pristine on
761
+ every lease (spine §4.2), and reconciles an unhealthy world back in via `provision()` again
762
+ (§4 rule 1: "a sick world mid-job is recovered by calling `provision` again") — in the
763
+ background, so a lease elsewhere never blocks on someone else's recovery.
764
+
765
+ B1/B2/M6 (spine v1.12 §4.5b): the provider port is NOT reentrant — at most one
766
+ `provision`/`reset`/`healthy`/`close` call is ever in flight, serialized by `_provider_lock`
767
+ (R13: `healthy` writes — it demotes state — so v1.12 folded it into the same serialized set
768
+ that provision/reset/close were already in; it is no longer treated as a read-only probe
769
+ exempt from the lock). A demotion that lands while a reconcile is already running is coalesced
770
+ into a trailing pass rather than a second concurrent `provision()` call.
771
+ """
772
+
773
+ def __init__(
774
+ self,
775
+ provisioner: WorldProvisioner,
776
+ *,
777
+ bundle: Any,
778
+ source: Path,
779
+ bundle_dir: Path,
780
+ work_directory: Path,
781
+ instances: int,
782
+ outbound: OutboundPort | None = None,
783
+ ) -> None:
784
+ self._provisioner = provisioner
785
+ self._bundle = bundle
786
+ self._source = source
787
+ self._bundle_dir = bundle_dir
788
+ self._work_directory = work_directory
789
+ self._instances = instances
790
+ self._outbound = outbound
791
+
792
+ self._runtimes: dict[int, EnvironmentRuntime] = {}
793
+ self._available: set[int] = set()
794
+ self._leased: set[int] = set()
795
+ self._down: set[int] = set()
796
+ self._fresh: set[int] = (
797
+ set()
798
+ ) # m9: provisioned/recovered but never yet leased/reset
799
+ self._effective_size = 0 # R2: the achieved world count `start()` settled on
800
+ # The §2f (code, domain) pair (or `None`) behind the most recent demotion/reconcile-failure
801
+ # for a down world index -- read by `lease()`'s exhaustion check to decide whether a
802
+ # uniform never-retried code can surface instead of the generic `world_pool_exhausted`.
803
+ # `domain` is the CARRIED value off the typed error (v1.15), captured once here rather than
804
+ # re-derived later from the code alone.
805
+ self._down_codes: dict[int, tuple[str, FailureDomain] | None] = {}
806
+ self._fenced: BaseException | None = (
807
+ None # latched by mark_fenced(), never cleared
808
+ )
809
+
810
+ # m1: `asyncio.Condition` (not a manual `Event` + `clear()`) — waiting and notifying share
811
+ # one lock, so there is no window between releasing a lock and clearing a flag for a
812
+ # `set()` to land in and be silently lost.
813
+ self._state_lock = asyncio.Condition()
814
+ self._provider_lock = asyncio.Lock()
815
+ self._reconcile_task: asyncio.Task[None] | None = None
816
+ self._reconcile_pending = False
817
+ self._started = False
818
+ self._closing = (
819
+ False # R4: set at the top of close() -- lets an in-flight reconcile bail
820
+ )
821
+ # between attempts instead of burning close()'s wait budget on a pool being torn down.
822
+ self._closed = (
823
+ False # Set once close() STARTS -- latches provision()/lease() out for
824
+ )
825
+ # good immediately, independent of whether teardown itself has finished.
826
+ self._teardown_task: asyncio.Task[None] | None = None # the shared, retry-safe
827
+ # teardown -- see close()'s own comment for why idempotency lives here now, not on
828
+ # `_closed`.
829
+
830
+ @property
831
+ def effective_size(self) -> int:
832
+ """R2: the world count `start()` actually achieved — may be less than the requested
833
+ `instances` on a legitimate degrade (conformance-gate failure, `fixed_port`). P10 sizes
834
+ `parallelism_degraded` and anything else that needs "how many worlds do we really have"
835
+ off this, never off the originally requested `instances`."""
836
+ return self._effective_size
837
+
838
+ @property
839
+ def fenced(self) -> BaseException | None:
840
+ """The first fatal `OutboundPort` exception (401/403 -> `HostedFencedError`, a
841
+ 404-exhausted channel -> `HostedChannelFailedError`, or a 409 attempt-supersession ->
842
+ `HostedAttemptSupersededError`) observed anywhere along this pool's own emit paths.
843
+ `HostedScheduler` polls this at the same points it polls `cancel_requested` to stop
844
+ leasing/launching further scenarios once set."""
845
+ return self._fenced
846
+
847
+ def mark_fenced(self, exc: BaseException) -> None:
848
+ if self._fenced is None:
849
+ self._fenced = exc
850
+
851
+ @property
852
+ def size(self) -> int:
853
+ return len(self._runtimes)
854
+
855
+ async def start(self) -> list[EnvironmentRuntime]:
856
+ if self._started:
857
+ # m10: a second call would re-provision behind every already-leased world's back.
858
+ raise RuntimeError("WorldPool.start() called more than once")
859
+ self._started = True
860
+
861
+ async with self._provider_lock:
862
+ runtimes = await self._provisioner.provision(
863
+ self._bundle,
864
+ source=self._source,
865
+ bundle_dir=self._bundle_dir,
866
+ work_directory=self._work_directory,
867
+ instances=self._instances,
868
+ )
869
+
870
+ # R2 (spine v1.12 §4's conformance gate / `fixed_port`): `provision()` legitimately
871
+ # returns FEWER than `instances` worlds — "Fail → effective parallelism 1 +
872
+ # parallelism_degraded ... Loud, never silent," not a failure this pool should raise on.
873
+ # Reject only a genuinely malformed result: zero worlds, duplicates, a non-contiguous
874
+ # index set, or more worlds than were ever requested.
875
+ indices = {runtime.world_index for runtime in runtimes}
876
+ if (
877
+ not runtimes
878
+ or len(runtimes) != len(indices)
879
+ or indices != set(range(len(runtimes)))
880
+ or len(runtimes) > self._instances
881
+ ):
882
+ # m10/R2: spine §4 — "ordered by world_index" and contiguous from 0 (what
883
+ # `range(effective_instances)` on the provider side guarantees).
884
+ raise RuntimeError(
885
+ f"provision() returned world_index set {sorted(indices)}, expected a contiguous "
886
+ f"0..N-1 subset of 0..{self._instances - 1}"
887
+ )
888
+ self._effective_size = len(runtimes)
889
+
890
+ async with self._state_lock:
891
+ for runtime in runtimes:
892
+ self._runtimes[runtime.world_index] = runtime
893
+ if runtime.state in (RuntimeState.READY, RuntimeState.PREPARING):
894
+ # m10: never hand out a world provision() itself returned UNHEALTHY. A
895
+ # PREPARING world legitimately demotes straight to UNHEALTHY on a failed first
896
+ # reset/probe (spine v1.12 §3's preparing->unhealthy transition) -- lease()'s
897
+ # own health gate covers that case; nothing extra is needed here.
898
+ self._available.add(runtime.world_index)
899
+ if runtime.state is RuntimeState.READY:
900
+ self._fresh.add(runtime.world_index)
901
+ else:
902
+ self._down.add(runtime.world_index)
903
+ self._state_lock.notify_all()
904
+ return runtimes
905
+
906
+ def _reconcile_in_flight(self) -> bool:
907
+ return self._reconcile_task is not None and not self._reconcile_task.done()
908
+
909
+ async def _wait_bounded(self, *, poll: bool) -> None:
910
+ if not poll:
911
+ await self._state_lock.wait()
912
+ return
913
+ try:
914
+ await asyncio.wait_for(
915
+ self._state_lock.wait(), timeout=_LEASE_POLL_INTERVAL_SECONDS
916
+ )
917
+ except asyncio.TimeoutError:
918
+ pass # `Condition.wait()` reacquires the lock before propagating even on timeout.
919
+
920
+ async def lease(
921
+ self,
922
+ *,
923
+ exclude: frozenset[int] = frozenset(),
924
+ abandon: Callable[[], bool] | None = None,
925
+ ) -> tuple[int, EnvironmentRuntime] | None:
926
+ """Returns `None` if `abandon()` reports true while this call was queued (B5) — the caller
927
+ never received a world, so there is nothing to release."""
928
+ while True:
929
+ # R5: latched once close() has run — a lease past that point must never spawn a
930
+ # `reset()`/`healthy()` call against a provider that may already be hard-cleaned.
931
+ if self._closed:
932
+ raise NoWorldsAvailable("world pool is closed", reason="closed")
933
+ if abandon is not None and abandon():
934
+ return None
935
+
936
+ async with self._state_lock:
937
+ candidates = self._available - exclude
938
+ if candidates:
939
+ world_index = min(candidates)
940
+ self._available.discard(world_index)
941
+ self._leased.add(world_index)
942
+ skip_reset = world_index in self._fresh # m9
943
+ self._fresh.discard(world_index)
944
+ else:
945
+ # Not just "every world is down" (the plain retry-exhausted case) — a world
946
+ # excluded for this lease (a same-scenario retry avoiding its failed world)
947
+ # can never satisfy `candidates` again no matter how long we wait, so it must
948
+ # count as unusable here too or a single-world pool's retry blocks forever.
949
+ usable = set(self._runtimes) - exclude
950
+ if not (usable - self._down):
951
+ # M9/R10 (spine v1.12 §5.4): declare exhaustion only once no reconcile is
952
+ # in flight or about to be — never on an instantaneous snapshot of world
953
+ # states. `_reconcile_pending` (set inside `mark_unhealthy`'s own critical
954
+ # section, R10) covers the gap between a demotion and its reconcile task
955
+ # actually existing.
956
+ if self._reconcile_in_flight() or self._reconcile_pending:
957
+ await self._wait_bounded(poll=abandon is not None)
958
+ continue
959
+ # v1.13 §5.4: a uniform §2f never-retried code across every
960
+ # currently-unhealthy world surfaces AS that code+domain; mixed codes, an
961
+ # unrecorded (non-§2f) cause, or any infrastructure-domain code all fall
962
+ # back to the generic `world_pool_exhausted` exactly as before. The
963
+ # uniformity set is `self._down` -- every unhealthy world in the pool, not
964
+ # just `usable` (runtimes minus this call's `exclude`) -- a world excluded
965
+ # because it is the scenario's own just-failed world is still part of "every
966
+ # unhealthy world" the spec means; narrowing to `usable` would let that
967
+ # excluded world's own (possibly untyped) failure escape the check entirely.
968
+ codes = {self._down_codes.get(index) for index in self._down}
969
+ code = domain = None
970
+ if len(codes) == 1:
971
+ (only,) = codes
972
+ if only is not None:
973
+ only_code, only_domain = only
974
+ if only_domain in _SECTION_2F_NEVER_RETRIED:
975
+ code, domain = only_code, only_domain
976
+ raise NoWorldsAvailable(
977
+ f"{len(self._down)}/{len(self._runtimes)} worlds unhealthy, "
978
+ f"none available outside {sorted(exclude)}",
979
+ code=code,
980
+ domain=domain,
981
+ )
982
+ await self._wait_bounded(poll=abandon is not None)
983
+ continue
984
+
985
+ reset_exc: Exception | None = None
986
+ probed_runtime: EnvironmentRuntime | None = None
987
+ if not skip_reset:
988
+ async with self._provider_lock:
989
+ if self._closed:
990
+ # close() can win the `_provider_lock` FIFO queue against a lease already
991
+ # past the top-of-loop `_closed` check -- re-check on the inside too, or
992
+ # this lease drives reset() against a provider close() may already be
993
+ # hard-cleaning.
994
+ raise NoWorldsAvailable("world pool is closed", reason="closed")
995
+ runtime = self._runtimes.get(world_index)
996
+ probed_runtime = runtime
997
+ if runtime is not None:
998
+ try:
999
+ await self._provisioner.reset(
1000
+ runtime, work_directory=self._work_directory
1001
+ )
1002
+ except Exception as exc: # noqa: BLE001 - B3: must never leak out of lease()
1003
+ reset_exc = exc
1004
+
1005
+ is_healthy = False
1006
+ if reset_exc is None:
1007
+ # M2: `healthy()` is called unconditionally after reset — including the m9 fast
1008
+ # path, which skips only the (expensive) reset call, never the readiness check.
1009
+ # R13 (spine v1.12 §4.5b): `healthy` now rides the port's non-reentrancy rule too,
1010
+ # so it goes under `_provider_lock` like reset/provision/close.
1011
+ async with self._provider_lock:
1012
+ if self._closed:
1013
+ raise NoWorldsAvailable(
1014
+ "world pool is closed", reason="closed"
1015
+ ) # same re-check as above
1016
+ runtime = self._runtimes.get(world_index)
1017
+ probed_runtime = runtime
1018
+ if runtime is not None:
1019
+ try:
1020
+ is_healthy = await self._provisioner.healthy(
1021
+ runtime, work_directory=self._work_directory
1022
+ )
1023
+ except Exception as exc: # noqa: BLE001
1024
+ reset_exc = exc
1025
+
1026
+ async with self._state_lock:
1027
+ # m2/R14: re-read after the awaited provider calls — a concurrent reconcile may
1028
+ # have replaced or dropped this index's `EnvironmentRuntime` while lease() awaited.
1029
+ # `is_healthy` was computed against `probed_runtime` specifically; if the object
1030
+ # at this index is no longer that same object, the verdict no longer describes it
1031
+ # — discard this attempt and let the outer loop re-evaluate the index fresh rather
1032
+ # than apply a stale verdict to a new object.
1033
+ runtime = self._runtimes.get(world_index)
1034
+ if runtime is None or runtime is not probed_runtime:
1035
+ self._leased.discard(world_index)
1036
+ if runtime is not None:
1037
+ # The object was REPLACED, not removed -- put the index back where the
1038
+ # outer loop can find it, or it lands nowhere (not available, not down)
1039
+ # and every future lease() spins forever on a candidate set that never
1040
+ # grows.
1041
+ self._available.add(world_index)
1042
+ self._state_lock.notify_all()
1043
+ continue
1044
+ if is_healthy and runtime.state is RuntimeState.READY:
1045
+ self._state_lock.notify_all()
1046
+ return world_index, runtime
1047
+ cause = (
1048
+ f"reset failed: {reset_exc}"
1049
+ if reset_exc is not None
1050
+ else f"reset left world in state {runtime.state.value}"
1051
+ )
1052
+ # Preserve a typed §2f code (and its CARRIED domain, v1.15) across this seam
1053
+ # instead of flattening it to free text -- `mark_unhealthy` records it so a later
1054
+ # exhaustion declaration can tell a deterministic never-retried fault apart from a
1055
+ # generic infrastructure one.
1056
+ is_typed = isinstance(reset_exc, ProcessRuntimeError)
1057
+ code = reset_exc.code if is_typed else None
1058
+ domain = reset_exc.domain if is_typed else None
1059
+
1060
+ await self.mark_unhealthy(
1061
+ world_index, cause=cause, code=code, domain=domain
1062
+ )
1063
+ # loop again — this index is now excluded via `_down`, no explicit retry bookkeeping.
1064
+
1065
+ async def release(self, world_index: int) -> None:
1066
+ async with self._state_lock:
1067
+ self._leased.discard(world_index)
1068
+ if world_index in self._runtimes and world_index not in self._down:
1069
+ self._available.add(world_index)
1070
+ self._state_lock.notify_all()
1071
+
1072
+ async def mark_unhealthy(
1073
+ self,
1074
+ world_index: int,
1075
+ *,
1076
+ cause: str,
1077
+ code: str | None = None,
1078
+ domain: FailureDomain | None = None,
1079
+ ) -> None:
1080
+ # `domain` is the CARRIED value off a typed provisioner error (v1.15); `None` here (e.g. a
1081
+ # caller that only has a bare code) falls back to the closed map via `_resolve_2f_domain`,
1082
+ # logged when it fires.
1083
+ async with self._state_lock:
1084
+ self._leased.discard(world_index)
1085
+ self._available.discard(world_index)
1086
+ self._fresh.discard(world_index)
1087
+ self._down.add(world_index)
1088
+ # Unconditional -- every demotion overwrites the recorded reason (or clears a stale
1089
+ # §2f code with `None` when this one isn't typed), so exhaustion always reads the
1090
+ # MOST RECENT cause for this index, never a leftover from an earlier failure.
1091
+ self._down_codes[world_index] = _resolve_2f_domain(code, domain)
1092
+ runtime = self._runtimes.get(world_index)
1093
+ if runtime is not None:
1094
+ # M12 (spine v1.12 §4.5b, normative): the scheduler demotes `state` on the
1095
+ # provider's own live `EnvironmentRuntime` object — that demotion is the signal
1096
+ # the NEXT `provision()` reconciles on.
1097
+ runtime.state = RuntimeState.UNHEALTHY
1098
+ # R10: set inside this same critical section (not left to `_schedule_reconcile`'s own,
1099
+ # later one) so a `lease()` observing state in the gap between the two never sees
1100
+ # "every world down, no reconcile in flight or pending" and raises spuriously.
1101
+ self._reconcile_pending = True
1102
+ self._state_lock.notify_all()
1103
+
1104
+ # Schedule recovery BEFORE the telemetry emit below -- `OutboundPort` calls are
1105
+ # best-effort and may be slow or hang, and recovery must never sit behind one (worst
1106
+ # case: `_reconcile_pending` stays latched and `lease()`'s grace loop spins forever).
1107
+ await self._schedule_reconcile()
1108
+
1109
+ # R6: this is the sole path every demotion (this method) goes through, so it is the one
1110
+ # place `world_unhealthy` needs to be emitted from for all four call sites to get it.
1111
+ if self._outbound is not None:
1112
+ try:
1113
+ await self._outbound.world_unhealthy(
1114
+ world_index=world_index, cause=_sanitize_cause(cause)
1115
+ )
1116
+ except (
1117
+ _FATAL_OUTBOUND
1118
+ ) as exc: # a fence stops the run -- never best-effort.
1119
+ self.mark_fenced(exc)
1120
+ except Exception as exc: # noqa: BLE001 - B3: outbound failures are never fatal.
1121
+ await self._log(f"world_unhealthy emit failed: {exc}")
1122
+
1123
+ async def _schedule_reconcile(self) -> None:
1124
+ async with self._state_lock:
1125
+ if self._closed:
1126
+ return # R5: never spawn new provider work once the pool has been closed.
1127
+ if self._reconcile_in_flight():
1128
+ # B1/M6: a demotion landing mid-reconcile is coalesced into a trailing pass
1129
+ # (`_reconcile_loop`) rather than a second concurrent `provision()` call.
1130
+ self._reconcile_pending = True
1131
+ return
1132
+ self._reconcile_task = asyncio.create_task(self._reconcile_loop())
1133
+
1134
+ async def _reconcile_loop(self) -> None:
1135
+ while True:
1136
+ async with self._state_lock:
1137
+ self._reconcile_pending = False
1138
+ await self._reconcile()
1139
+ async with self._state_lock:
1140
+ if not self._reconcile_pending:
1141
+ return
1142
+
1143
+ async def _reconcile(self) -> None:
1144
+ # M5: bounded retry with backoff — a single transient `provision()` failure (a momentary
1145
+ # ENOSPC, an engine hiccup) used to retire its world for the rest of the job with no
1146
+ # signal anywhere. Every failed attempt is logged through `OutboundPort` (when wired),
1147
+ # matching the contract's own "loud, never silent" standard for degradation.
1148
+ runtimes: list[EnvironmentRuntime] | None = None
1149
+ last_exc: Exception | None = None
1150
+ for attempt in range(1, _RECONCILE_MAX_ATTEMPTS + 1):
1151
+ if self._closing:
1152
+ # R4: close() is already bounded-waiting on this task — do not spend its wait
1153
+ # budget retrying a pool that is being torn down anyway.
1154
+ return
1155
+ try:
1156
+ async with self._provider_lock:
1157
+ runtimes = await self._provisioner.provision(
1158
+ self._bundle,
1159
+ source=self._source,
1160
+ bundle_dir=self._bundle_dir,
1161
+ work_directory=self._work_directory,
1162
+ instances=self._instances,
1163
+ )
1164
+ except Exception as exc: # noqa: BLE001 - a reconcile must never crash the pool
1165
+ last_exc = exc
1166
+ await self._log(
1167
+ f"world pool reconcile attempt {attempt}/{_RECONCILE_MAX_ATTEMPTS} failed: {exc}"
1168
+ )
1169
+ if attempt < _RECONCILE_MAX_ATTEMPTS and not self._closing:
1170
+ await asyncio.sleep(_RECONCILE_BACKOFF_SECONDS[attempt - 1])
1171
+ continue
1172
+ last_exc = None
1173
+ break
1174
+
1175
+ if last_exc is not None or runtimes is None:
1176
+ # The FINAL failed re-provision attempt's typed §2f code (and its CARRIED domain,
1177
+ # v1.15), applied to every world still down when this reconcile gives up -- one
1178
+ # `provision()` call covers the whole pool, so a typed failure here is uniform by
1179
+ # construction across everything it did not just recover.
1180
+ is_typed = isinstance(last_exc, ProcessRuntimeError)
1181
+ code_and_domain = _resolve_2f_domain(
1182
+ last_exc.code if is_typed else None,
1183
+ last_exc.domain if is_typed else None,
1184
+ )
1185
+ # R8: every success path below ends in `notify_all()` — this give-up path must too,
1186
+ # or a `lease()` blocked in `_wait_bounded(poll=False)` (the `abandon is None` case)
1187
+ # waits forever for a reconcile that already gave up.
1188
+ # Unconditional, mirroring `mark_unhealthy`'s own invariant -- an untyped final
1189
+ # attempt must overwrite (clear) a stale typed code left by an earlier demotion, or
1190
+ # exhaustion later reads that leftover code as if it were this attempt's own result.
1191
+ async with self._state_lock:
1192
+ for index in self._down:
1193
+ self._down_codes[index] = code_and_domain
1194
+ self._state_lock.notify_all()
1195
+ return # stays `_down`; the next `mark_unhealthy` (or a lease-triggered wait) retries.
1196
+
1197
+ # M12: recovery is judged by re-probing `healthy()` (M2's port), never by reading `state`
1198
+ # back — the scheduler is what wrote `state` when it demoted this world, so trusting it
1199
+ # here would be reading our own signal as independent proof. R13 (spine v1.12 §4.5b):
1200
+ # `healthy` now rides the port's non-reentrancy rule, so these probes go under
1201
+ # `_provider_lock` too.
1202
+ if self._closing:
1203
+ # provision() just succeeded, but close() may already be queued on `_provider_lock`
1204
+ # for its own `provisioner.close()` call -- bail before racing it for one more round
1205
+ # of provider calls the pool is being torn down under anyway.
1206
+ return
1207
+ healthy_by_index: dict[int, bool] = {}
1208
+ # The probe's own §2f code, carried alongside its verdict -- a world that comes back from a
1209
+ # SUCCESSFUL `provision()` but fails this probe never enters the give-up path above (that
1210
+ # path only fires on a raised/failed `provision()`), so without this the state block below
1211
+ # has no code of its own and would otherwise leave whatever an earlier, superseded demotion
1212
+ # recorded standing.
1213
+ healthy_codes: dict[int, tuple[str, FailureDomain] | None] = {}
1214
+ async with self._provider_lock:
1215
+ for runtime in runtimes:
1216
+ try:
1217
+ healthy_by_index[
1218
+ runtime.world_index
1219
+ ] = await self._provisioner.healthy(
1220
+ runtime, work_directory=self._work_directory
1221
+ )
1222
+ healthy_codes[runtime.world_index] = None
1223
+ except Exception as exc: # noqa: BLE001
1224
+ healthy_by_index[runtime.world_index] = False
1225
+ is_typed = isinstance(exc, ProcessRuntimeError)
1226
+ healthy_codes[runtime.world_index] = _resolve_2f_domain(
1227
+ exc.code if is_typed else None,
1228
+ exc.domain if is_typed else None,
1229
+ )
1230
+
1231
+ achieved = {runtime.world_index for runtime in runtimes}
1232
+ async with self._state_lock:
1233
+ for runtime in runtimes:
1234
+ self._runtimes[runtime.world_index] = runtime
1235
+ if healthy_by_index.get(runtime.world_index, False):
1236
+ was_down = runtime.world_index in self._down
1237
+ self._down.discard(runtime.world_index)
1238
+ self._down_codes.pop(
1239
+ runtime.world_index, None
1240
+ ) # recovered -- stale now
1241
+ if runtime.world_index not in self._leased:
1242
+ self._available.add(runtime.world_index)
1243
+ if was_down and runtime.state is RuntimeState.READY:
1244
+ self._fresh.add(runtime.world_index) # m9
1245
+ elif runtime.world_index in self._down:
1246
+ # Still down after a successful re-provision -- this probe's own result
1247
+ # replaces whatever an earlier demotion left, never a leftover from before it.
1248
+ self._down_codes[runtime.world_index] = healthy_codes.get(
1249
+ runtime.world_index
1250
+ )
1251
+ # `provision` reconciles to exactly `instances` worlds (a conformance-gate degrade can
1252
+ # shrink `achieved` below what this pool started with) — anything no longer returned
1253
+ # is gone, not merely unhealthy.
1254
+ for stale in [index for index in self._runtimes if index not in achieved]:
1255
+ self._runtimes.pop(stale, None)
1256
+ self._available.discard(stale)
1257
+ self._down.discard(stale)
1258
+ self._fresh.discard(stale)
1259
+ self._down_codes.pop(stale, None) # the index itself is gone
1260
+ # m3: NOT `_leased.discard(stale)` — an in-flight scenario may still hold this
1261
+ # index's lease (e.g. a conformance degrade shrinking `achieved` mid-scenario);
1262
+ # dropping the lease record here would make its later `release()`/
1263
+ # `mark_unhealthy()` a silent no-op. Those methods already guard on
1264
+ # `world_index in self._runtimes`, so leaving `_leased` alone and letting them
1265
+ # reconcile it lazily is correct.
1266
+ # Keep this truthful across a reconcile, not just at start() -- P10 sizes
1267
+ # `parallelism_degraded` off it, and a reconcile can grow the pool back up or shrink
1268
+ # it further (a conformance degrade narrowing `achieved`) in either direction.
1269
+ self._effective_size = len(self._runtimes)
1270
+ self._state_lock.notify_all()
1271
+
1272
+ async def close(self) -> None:
1273
+ async with self._state_lock:
1274
+ if not self._closed:
1275
+ self._closed = True
1276
+ self._closing = True
1277
+ # Wake anything blocked in `lease()`'s `_wait_bounded(poll=False)` so it
1278
+ # re-checks `_closed` instead of waiting for a recovery that will never come.
1279
+ self._state_lock.notify_all()
1280
+ # The OLD idempotency check (`if self._closed: return`) latched here, before teardown
1281
+ # ever ran -- a caller wrapping this whole call in its own timeout (the entrypoint's
1282
+ # `_bounded_close`) could cancel it mid-teardown, and a RETRY then hit that early
1283
+ # return and silently never called `provisioner.close()` at all. `_closed` still has
1284
+ # to latch immediately (lease()'s top-of-loop check, its inner re-check under
1285
+ # `_provider_lock`, and `_teardown`'s own bail-out below all depend on new
1286
+ # leases/reconciles being rejected the moment close() STARTS, not once it finishes),
1287
+ # so idempotency now lives on a separate, SHARED teardown task instead: every call —
1288
+ # first or retried — creates it once and awaits the same one.
1289
+ if self._teardown_task is None:
1290
+ self._teardown_task = asyncio.create_task(self._teardown())
1291
+ teardown_task = self._teardown_task
1292
+
1293
+ # `asyncio.shield`: if THIS call's own awaiter is cancelled (the caller's timeout fires),
1294
+ # the cancellation stops at this `await` and never reaches `teardown_task` -- teardown
1295
+ # keeps running in the background, and a retry's `close()` re-attaches to the same
1296
+ # (possibly by-then-finished) task instead of no-op'ing.
1297
+ await asyncio.shield(teardown_task)
1298
+
1299
+ async def _teardown(self) -> None:
1300
+ async with self._state_lock:
1301
+ task = self._reconcile_task
1302
+ if task is not None:
1303
+ # §4.5b: do NOT cancel-then-close.
1304
+ # `ProcessRuntimeProvider.provision`/`reset`/`healthy` are `asyncio.to_thread` —
1305
+ # cancelling the awaiting coroutine does NOT stop the underlying thread, so the old
1306
+ # bounded-wait-then-cancel let `provisioner.close()` run CONCURRENTLY with a
1307
+ # still-live `provision()` once the bound expired: unsynchronized identity dicts
1308
+ # (`RuntimeError: dictionary changed size during iteration`), leaked engines, and
1309
+ # `close()` itself could raise out of the guest's terminal path. `_closing` (in
1310
+ # `_reconcile`'s own retry loop and healthy-probe gate) already makes a reconcile bail
1311
+ # BETWEEN attempts/probes without a cancel, so this waits for the ONE call already in
1312
+ # flight to finish on its own — unbounded from this function's perspective, but
1313
+ # bounded in practice by whichever single `provision()`/`healthy()` call was running,
1314
+ # with the outer flush-window deadline (spine, P10-owned) as the real backstop --
1315
+ # there is no longer a single constant here that bounds this wait on its own.
1316
+ await asyncio.gather(task, return_exceptions=True)
1317
+
1318
+ async with self._provider_lock:
1319
+ await self._provisioner.close(work_directory=self._work_directory)
1320
+
1321
+ async def _log(self, message: str, *, level: str = "error") -> None:
1322
+ if self._outbound is None:
1323
+ return
1324
+ try:
1325
+ # R9: reuses `world_unhealthy.cause`'s own sanitizer — a `provision()` failure
1326
+ # routinely carries a postgres error string with the DSN, and outbound-channels.md
1327
+ # requires redaction (no endpoint userinfo) before anything crosses the wire.
1328
+ await self._outbound.log(level=level, message=_sanitize_cause(message))
1329
+ except _FATAL_OUTBOUND as exc: # a fence stops the run -- never best-effort.
1330
+ self.mark_fenced(exc)
1331
+ except Exception: # noqa: BLE001 - B3: outbound failures are never fatal.
1332
+ pass
1333
+
1334
+
1335
+ # --- the scenario loop --------------------------------------------------------------------------
1336
+
1337
+
1338
+ @dataclass(frozen=True)
1339
+ class RunResult:
1340
+ """`receipts` mixes already-emitted real receipts with synthesized-but-not-yet-emitted
1341
+ `skipped` ones (R6) — see `HostedScheduler.emit_skipped_receipts`.
1342
+
1343
+ `fenced` is set once a 401/403 (`HostedFencedError`), a 404-exhausted channel
1344
+ (`HostedChannelFailedError`), or a 409 attempt-supersession (`HostedAttemptSupersededError`)
1345
+ was observed on any outbound call -- the run stops launching further scenarios the moment it is
1346
+ set. The caller maps this to exit code 3 and must not call `emit_skipped_receipts` (no further
1347
+ outbound emission once fenced)."""
1348
+
1349
+ receipts: tuple[ResultReceipt, ...]
1350
+ aborted: ReceiptFailure | None
1351
+ fenced: BaseException | None = None
1352
+
1353
+
1354
+ def _skipped_receipt(scenario: Scenario) -> ResultReceipt:
1355
+ # Exact body per outbound-channels.md Channel 2, "skipped receipt body (exact)".
1356
+ return ResultReceipt(
1357
+ scenario_key=scenario.scenario_key,
1358
+ scenario_id=scenario.scenario_id,
1359
+ scenario_attempt=1,
1360
+ world_index=None,
1361
+ status="skipped",
1362
+ sub_goals=(),
1363
+ evaluations=(),
1364
+ call=None,
1365
+ failure=None,
1366
+ )
1367
+
1368
+
1369
+ def _unjudged(sub_goals: Sequence[SubGoal]) -> tuple[SubGoalResult, ...]:
1370
+ return tuple(
1371
+ # R11: outbound-channels.md pins `judged` as `SubGoal.judged != ""`, not `bool(...)` —
1372
+ # they agree for every `str` but `bool` is not what the contract names.
1373
+ SubGoalResult(name=goal.name, held=None, reason=None, judged=goal.judged != "")
1374
+ for goal in sub_goals
1375
+ )
1376
+
1377
+
1378
+ _LEAK_HEADROOM = (
1379
+ 10 # R1: spine §1's hosted `scenario_count` admission range is 1..10 -- the most
1380
+ )
1381
+ # phase threads that can ever be simultaneously abandoned (leaked) in one job.
1382
+
1383
+
1384
+ def _abort_from_no_worlds(exc: NoWorldsAvailable) -> ReceiptFailure:
1385
+ # A uniform §2f never-retried code across every unhealthy world surfaces AS that code+domain;
1386
+ # otherwise this is the generic exhaustion abort.
1387
+ if exc.code is not None and exc.domain is not None:
1388
+ return ReceiptFailure(
1389
+ domain=exc.domain.value,
1390
+ stage=HarnessStage.RUNNING.value,
1391
+ code=exc.code,
1392
+ message=_truncate(str(exc)),
1393
+ )
1394
+ return _failure("world_pool_exhausted", str(exc))
1395
+
1396
+
1397
+ @dataclass
1398
+ class _ScenarioContext:
1399
+ """R7: `_run_scenario` records the world/attempt it is currently working on here as it goes,
1400
+ so a crash that escapes every handled path still lets `worker()` report the REAL
1401
+ world_index/scenario_attempt on the `driver_crashed` receipt instead of always None/1.
1402
+ `call` is set the moment the call step returns, so a LATER crash (e.g. `read_only()`
1403
+ building the check-phase handle) still reports the call that genuinely ran, not `null`."""
1404
+
1405
+ world_index: int | None = None
1406
+ attempt: int = 1
1407
+ call: CallSummary | None = None
1408
+
1409
+
1410
+ @dataclass(frozen=True)
1411
+ class _PendingRetryReceipt:
1412
+ """R3: carries attempt-1's already-built `_Retry` outcome across the retry-lease boundary, so
1413
+ a cancel/abort landing anywhere between "attempt 1 finished" and "attempt 2 actually starts"
1414
+ still reports what attempt 1 produced instead of losing it to skipped-synthesis (the same
1415
+ defect M8 fixed on the `NoWorldsAvailable` branch, on the other post-attempt-1 exit)."""
1416
+
1417
+ world_index: int
1418
+ attempt: int
1419
+ outcome: "_Retry"
1420
+
1421
+
1422
+ def _record_scenario(span: Any, receipt: Any, context: Any) -> None:
1423
+ """Put the scenario's verdict on its span, so a trace answers what happened without a receipt."""
1424
+ if span is None or receipt is None:
1425
+ return
1426
+ failure = getattr(receipt, "failure", None)
1427
+ observability.record(
1428
+ span,
1429
+ status=getattr(receipt, "status", None),
1430
+ failure_code=getattr(failure, "code", None),
1431
+ failure_domain=getattr(failure, "domain", None),
1432
+ world_index=getattr(context, "world_index", None),
1433
+ attempt=getattr(context, "attempt", None),
1434
+ )
1435
+
1436
+ class HostedScheduler:
1437
+ """Drains a job's scenario list across a `WorldPool`, one asyncio task per scenario — lease()
1438
+ blocking when the pool is saturated is what caps concurrency at W, so nothing here re-derives
1439
+ a worker count. Retry is fixed at one extra attempt on a fresh world (spine §5 step 4), gated
1440
+ on `FailureDomain` per the P9 brief: retryable domains retry once, deterministic ones do not.
1441
+ """
1442
+
1443
+ def __init__(
1444
+ self,
1445
+ *,
1446
+ pool: WorldPool,
1447
+ world_factory: WorldFactory,
1448
+ call_runner: CallRunner,
1449
+ outbound: OutboundPort,
1450
+ job_seed: int,
1451
+ cancel_requested: Callable[[], bool] | None = None,
1452
+ judge: JudgeFn | None = None,
1453
+ ) -> None:
1454
+ self._pool = pool
1455
+ self._world_factory = world_factory
1456
+ self._call_runner = call_runner
1457
+ self._outbound = outbound
1458
+ self._job_seed = job_seed
1459
+ self._cancel_requested = cancel_requested or (lambda: False)
1460
+ # Injected like every other collaborator, so a test decides a judged sub-goal without a
1461
+ # model call. Resolved here rather than as a default argument, which would bind at import
1462
+ # and ignore both injection and patching.
1463
+ self._judge = judge or _judge
1464
+ self._executor: ThreadPoolExecutor | None = None
1465
+
1466
+ async def run(self, scenarios: Sequence[Scenario]) -> RunResult:
1467
+ results: list[ResultReceipt | None] = [None] * len(scenarios)
1468
+ abort_holder: list[ReceiptFailure | None] = [None]
1469
+
1470
+ # R1: a dedicated executor for scenario phase threads — never the loop's default
1471
+ # executor, which the provider's own `to_thread` calls (process_runtime.py) also use, and
1472
+ # whose capacity a leaked phase thread would starve globally. One worker per live world
1473
+ # plus headroom for the worst case of every admitted scenario leaking its own abandoned
1474
+ # thread at once (world-handle-interface.md: "its thread leaks, bounded by scenario
1475
+ # count").
1476
+ self._executor = ThreadPoolExecutor(
1477
+ # `_LEAK_HEADROOM` alone assumes spine §1's `scenario_count` admission cap (<=10) —
1478
+ # widen for whatever `scenarios` actually holds, or an over-cap job's overflow
1479
+ # scenarios find the executor saturated and report `driver_crashed` for a phase that
1480
+ # was queued, not run.
1481
+ max_workers=max(
1482
+ self._pool.effective_size + _LEAK_HEADROOM, len(scenarios) + 1
1483
+ ),
1484
+ thread_name_prefix="hosted-scenario",
1485
+ )
1486
+ try:
1487
+
1488
+ async def worker(index: int, scenario: Scenario) -> None:
1489
+ # `self._pool.fenced` is the same stop-path as `abort_holder`/`cancel_requested`
1490
+ # -- once any outbound call has hit a 401/403 or an exhausted channel, no further
1491
+ # scenario may even start.
1492
+ if (
1493
+ abort_holder[0] is not None
1494
+ or self._pool.fenced is not None
1495
+ or self._cancel_requested()
1496
+ ):
1497
+ return
1498
+ context = _ScenarioContext()
1499
+ try:
1500
+ with observability.scenario(
1501
+ str(getattr(scenario, "key", "") or index), index
1502
+ ) as span:
1503
+ results[index] = await self._run_scenario(
1504
+ scenario, index, abort_holder=abort_holder, context=context
1505
+ )
1506
+ _record_scenario(span, results[index], context)
1507
+ except NoWorldsAvailable as exc:
1508
+ abort_holder[0] = _abort_from_no_worlds(exc)
1509
+ except _FATAL_OUTBOUND:
1510
+ # Already latched onto `self._pool.fenced` by whichever `_emit`/`_log` call
1511
+ # raised it -- no receipt for a scenario the platform already superseded.
1512
+ pass
1513
+ except asyncio.CancelledError:
1514
+ raise
1515
+ except BaseException as exc: # noqa: BLE001
1516
+ # B3: the scheduler's own machinery crashing must not suppress every other
1517
+ # scenario's receipt — `gather(return_exceptions=True)` below is the second
1518
+ # half of that guarantee.
1519
+ results[index] = await self._driver_crashed_receipt(
1520
+ scenario,
1521
+ exc,
1522
+ world_index=context.world_index,
1523
+ scenario_attempt=context.attempt,
1524
+ call=context.call,
1525
+ )
1526
+
1527
+ tasks = [asyncio.create_task(worker(i, s)) for i, s in enumerate(scenarios)]
1528
+ if tasks:
1529
+ await asyncio.gather(*tasks, return_exceptions=True)
1530
+
1531
+ receipts: list[ResultReceipt] = []
1532
+ for index, scenario in enumerate(scenarios):
1533
+ receipt = results[index]
1534
+ if receipt is None:
1535
+ # (outbound-channels.md v1.3 Sequencing: "terminal event -> skipped
1536
+ # receipts -> manifest"): only SYNTHESIZE here. `run()` returns before its
1537
+ # caller has emitted a terminal event, so pushing this over `outbound` now
1538
+ # would put it on the wire ahead of the terminal -- `emit_skipped_receipts()`
1539
+ # is the caller's job, done AFTER its own terminal event.
1540
+ receipt = _skipped_receipt(scenario)
1541
+ receipts.append(receipt)
1542
+ return RunResult(
1543
+ receipts=tuple(receipts),
1544
+ aborted=abort_holder[0],
1545
+ fenced=self._pool.fenced,
1546
+ )
1547
+ finally:
1548
+ # R1: never block `run()` on abandoned threads — `shutdown(wait=True)` would hang
1549
+ # this coroutine exactly like the bug this fixes. Queued-but-unstarted work is
1550
+ # cancelled; already-running (leaked) threads are the contract's own accepted,
1551
+ # bounded tradeoff (world-handle-interface.md's "the job TTL is the backstop").
1552
+ self._executor.shutdown(wait=False, cancel_futures=True)
1553
+
1554
+ async def emit_skipped_receipts(self, result: RunResult) -> None:
1555
+ """R6: outbound-channels.md v1.3 Sequencing — "terminal event -> skipped receipts ->
1556
+ manifest". `run()` only synthesizes `skipped` receipts into `RunResult.receipts`; call
1557
+ this AFTER the caller's own terminal event has been emitted, never before, and exactly
1558
+ once — each call re-emits every `skipped` receipt in `result.receipts` with no dedup of
1559
+ its own.
1560
+
1561
+ A no-op once `self._pool.fenced` is set (checked live, so it also covers a fence that
1562
+ landed after `run()` returned but before this call) — the run stopped emitting the moment
1563
+ the fence was observed and must not resume for these. If a fence instead lands DURING this
1564
+ method's own loop, the same `_FATAL_OUTBOUND` that stops `run()` escapes out of this method
1565
+ too; the caller must be ready for that."""
1566
+ if self._pool.fenced is not None:
1567
+ return
1568
+ for receipt in result.receipts:
1569
+ if receipt.status == "skipped":
1570
+ await self._emit(self._outbound.receipt(receipt), what="receipt")
1571
+
1572
+ async def _emit(self, awaitable: Awaitable[None], *, what: str) -> None:
1573
+ # B3: `OutboundPort` exceptions are best-effort telemetry — never receipt-affecting and
1574
+ # never fatal to the run. Logged through the same port when logging itself doesn't also
1575
+ # fail; swallowed otherwise rather than let a transport hiccup kill the scenario loop.
1576
+ # The one exception besides `CancelledError` this deliberately does NOT swallow — a
1577
+ # fence (401/403) or an exhausted channel (404x3) is never best-effort telemetry.
1578
+ try:
1579
+ await awaitable
1580
+ except _FATAL_OUTBOUND as exc:
1581
+ self._pool.mark_fenced(exc)
1582
+ raise
1583
+ except Exception as exc: # noqa: BLE001
1584
+ try:
1585
+ await self._outbound.log(
1586
+ level="error", message=f"outbound.{what} failed: {exc}"
1587
+ )
1588
+ except _FATAL_OUTBOUND as log_exc:
1589
+ self._pool.mark_fenced(log_exc)
1590
+ raise
1591
+ except Exception: # noqa: BLE001
1592
+ pass
1593
+
1594
+ async def _driver_crashed_receipt(
1595
+ self,
1596
+ scenario: Scenario,
1597
+ exc: BaseException,
1598
+ *,
1599
+ world_index: int | None,
1600
+ scenario_attempt: int,
1601
+ call: CallSummary | None = None,
1602
+ ) -> ResultReceipt:
1603
+ failure = _failure("driver_crashed", f"{type(exc).__name__}: {exc}")
1604
+ try:
1605
+ # R7: best-effort — every declared goal, `held: null`, matching the errored-receipt
1606
+ # body's rule. Falls back to `()` only when reading `sub_goals` itself is what crashed
1607
+ # (the one case with no goal list to report at all).
1608
+ sub_goals = _unjudged(scenario.sub_goals)
1609
+ except Exception: # noqa: BLE001
1610
+ sub_goals = ()
1611
+ receipt = ResultReceipt(
1612
+ scenario_key=scenario.scenario_key,
1613
+ scenario_id=scenario.scenario_id,
1614
+ scenario_attempt=scenario_attempt,
1615
+ world_index=world_index,
1616
+ status="errored",
1617
+ sub_goals=sub_goals,
1618
+ evaluations=(),
1619
+ call=call, # the call step's own summary, if it had already returned when this crashed
1620
+ failure=failure,
1621
+ )
1622
+ await self._emit(self._outbound.receipt(receipt), what="receipt")
1623
+ return receipt
1624
+
1625
+ async def _emit_pending_retry_receipt(
1626
+ self, scenario: Scenario, pending: "_PendingRetryReceipt"
1627
+ ) -> ResultReceipt:
1628
+ # R3: the single shape both post-attempt-1 "never got to run attempt 2" exits emit.
1629
+ receipt = ResultReceipt(
1630
+ scenario_key=scenario.scenario_key,
1631
+ scenario_id=scenario.scenario_id,
1632
+ scenario_attempt=pending.attempt,
1633
+ world_index=pending.world_index,
1634
+ status="errored",
1635
+ sub_goals=pending.outcome.sub_goals,
1636
+ evaluations=(),
1637
+ call=pending.outcome.call,
1638
+ failure=pending.outcome.failure,
1639
+ )
1640
+ await self._emit(self._outbound.receipt(receipt), what="receipt")
1641
+ return receipt
1642
+
1643
+ async def _lease_or_abandon(
1644
+ self, *, exclude: frozenset[int], abort_holder: list[ReceiptFailure | None]
1645
+ ) -> tuple[int, EnvironmentRuntime] | None:
1646
+ def _abandon() -> bool:
1647
+ # A scenario already queued in `lease()` must also abandon once fenced -- the
1648
+ # worker-top check alone only stops scenarios that had not started yet.
1649
+ return (
1650
+ abort_holder[0] is not None
1651
+ or self._pool.fenced is not None
1652
+ or self._cancel_requested()
1653
+ )
1654
+
1655
+ return await self._pool.lease(exclude=exclude, abandon=_abandon)
1656
+
1657
+ async def _run_scenario(
1658
+ self,
1659
+ scenario: Scenario,
1660
+ scenario_index: int,
1661
+ *,
1662
+ attempt: int = 1,
1663
+ tried: frozenset[int] = frozenset(),
1664
+ pre_leased: tuple[int, EnvironmentRuntime] | None = None,
1665
+ pending_retry: "_PendingRetryReceipt | None" = None,
1666
+ abort_holder: list[ReceiptFailure | None],
1667
+ context: "_ScenarioContext",
1668
+ ) -> ResultReceipt | None:
1669
+ if pre_leased is not None:
1670
+ world_index, runtime = pre_leased
1671
+ else:
1672
+ leased = await self._lease_or_abandon(
1673
+ exclude=tried, abort_holder=abort_holder
1674
+ )
1675
+ if leased is None:
1676
+ return None # B5: cancelled/aborted while queued — never got a world
1677
+ world_index, runtime = leased
1678
+
1679
+ context.world_index = (
1680
+ world_index # R7: the real values for a driver_crashed receipt
1681
+ )
1682
+ context.attempt = attempt
1683
+ context.call = (
1684
+ None # this attempt has not made its own call yet -- must not still
1685
+ )
1686
+ # carry a previous attempt's summary on the shared context object into this one's receipt.
1687
+
1688
+ # B5: re-check immediately after `lease()` returns — a cancel/abort landing while this
1689
+ # worker was queued must not let a freshly granted world start work it can never finish
1690
+ # inside the flush window.
1691
+ if abort_holder[0] is not None or self._cancel_requested():
1692
+ await self._pool.release(world_index)
1693
+ if pending_retry is not None:
1694
+ # R3: attempt 1 already ran on `pending_retry.world_index` and produced a real
1695
+ # outcome — this is the retry continuation (this world was never used for it).
1696
+ return await self._emit_pending_retry_receipt(scenario, pending_retry)
1697
+ return None
1698
+
1699
+ world_resolved = (
1700
+ False # B3: the leased world must be released/discarded exactly once
1701
+ )
1702
+ try:
1703
+ if pending_retry is not None:
1704
+ # Emitted here, immediately before attempt 2's own `scenario_started`, so this
1705
+ # event and the pending-retry receipt (both exits above) are mutually exclusive
1706
+ # by construction — outbound-channels.md Channel 2: "the failed first try is
1707
+ # recorded by scenario_retried/world_unhealthy events, never by a receipt."
1708
+ await self._emit(
1709
+ self._outbound.scenario_retried(
1710
+ scenario_key=scenario.scenario_key,
1711
+ from_world=pending_retry.world_index,
1712
+ to_world=world_index,
1713
+ ),
1714
+ what="scenario_retried",
1715
+ )
1716
+ await self._emit(
1717
+ self._outbound.scenario_started(
1718
+ scenario_key=scenario.scenario_key,
1719
+ world_index=world_index,
1720
+ scenario_attempt=attempt,
1721
+ ),
1722
+ what="scenario_started",
1723
+ )
1724
+
1725
+ rng = random.Random(self._job_seed + scenario_index)
1726
+ outcome: ResultReceipt | _Retry
1727
+ try:
1728
+ world = await self._world_factory.create(runtime, rng=rng)
1729
+ except Exception as exc: # noqa: BLE001
1730
+ # B3: `world_factory.create()` failing (e.g. a PostgresStore connect failure) is
1731
+ # the same shape as a mid-scenario `WorldUnavailable` — the world is unusable, not
1732
+ # the scenario code. Deliberately narrow to just this call: `_execute()` has its
1733
+ # own exhaustive internal exception handling (`_run_phase`/`_invoke`), so anything
1734
+ # that still escapes it is a genuine scheduler bug and belongs in `driver_crashed`
1735
+ # (via `worker()`'s `BaseException` catch), not swallowed into `world_unavailable`.
1736
+ outcome = _Retry(
1737
+ _failure("world_unavailable", f"{type(exc).__name__}: {exc}"),
1738
+ sub_goals=_unjudged(scenario.sub_goals),
1739
+ call=None,
1740
+ mark_unhealthy=True,
1741
+ )
1742
+ else:
1743
+ outcome = await self._execute(
1744
+ scenario,
1745
+ world,
1746
+ runtime,
1747
+ world_index,
1748
+ attempt=attempt,
1749
+ context=context,
1750
+ )
1751
+
1752
+ if isinstance(outcome, _Retry):
1753
+ # R6: `mark_unhealthy()` itself emits `world_unhealthy` now (every demotion path
1754
+ # goes through it) — no separate emit needed here.
1755
+ if outcome.mark_unhealthy:
1756
+ await self._pool.mark_unhealthy(
1757
+ world_index, cause=outcome.failure.message
1758
+ )
1759
+ else:
1760
+ await self._pool.release(world_index)
1761
+ world_resolved = True
1762
+
1763
+ if attempt >= 2:
1764
+ receipt = ResultReceipt(
1765
+ scenario_key=scenario.scenario_key,
1766
+ scenario_id=scenario.scenario_id,
1767
+ scenario_attempt=attempt,
1768
+ world_index=world_index,
1769
+ status="errored",
1770
+ sub_goals=outcome.sub_goals,
1771
+ evaluations=(),
1772
+ call=outcome.call,
1773
+ failure=outcome.failure,
1774
+ )
1775
+ await self._emit(self._outbound.receipt(receipt), what="receipt")
1776
+ return receipt
1777
+
1778
+ # R3: attempt 1's outcome, carried forward so either exit below that never gets to
1779
+ # start attempt 2 can still report it instead of losing it to skipped-synthesis.
1780
+ pending = _PendingRetryReceipt(
1781
+ world_index=world_index, attempt=attempt, outcome=outcome
1782
+ )
1783
+ # A retry normally moves to another world. With an effective one-world pool,
1784
+ # excluding the only world makes the promised retry impossible: lease() raises
1785
+ # NoWorldsAvailable even though release() has returned a healthy runtime and the
1786
+ # next lease will reset it. Reuse is safe for non-unhealthy failures because the
1787
+ # lease path always resets the world before attempt 2. An unhealthy world is
1788
+ # still demoted and therefore cannot be leased until reconciliation replaces it.
1789
+ retry_exclude = tried | {world_index}
1790
+ if self._pool.effective_size == 1:
1791
+ retry_exclude = frozenset()
1792
+ try:
1793
+ next_leased = await self._lease_or_abandon(
1794
+ exclude=retry_exclude, abort_holder=abort_holder
1795
+ )
1796
+ except NoWorldsAvailable as exc:
1797
+ # M8: this scenario already ran and produced a real attempt-1 failure — losing
1798
+ # it to skipped-synthesis just because the retry lease found nothing would
1799
+ # report "never ran" for a scenario that manifestly did.
1800
+ #
1801
+ # P=1 failure preservation (v1.15): when the pool exhaustion carries no typed
1802
+ # §2f code of its own (exc.code is None), the original call failure is more
1803
+ # informative than the generic world_pool_exhausted — preserve it as the
1804
+ # job-level abort so a deterministic call_failed/CallAborted is returned as
1805
+ # its original typed failure, not masked behind world_pool_exhausted.
1806
+ # The scenario already produced the causative typed failure. A subsequent
1807
+ # inability to allocate its retry world must not replace that evidence with a
1808
+ # generic pool/provisioning wrapper (the masking seen in the hosted voice
1809
+ # timeout run).
1810
+ if pending.outcome.failure is not None:
1811
+ abort_holder[0] = pending.outcome.failure
1812
+ else:
1813
+ abort_holder[0] = _abort_from_no_worlds(exc) # v1.13 §5.4
1814
+ return await self._emit_pending_retry_receipt(scenario, pending)
1815
+
1816
+ if next_leased is None:
1817
+ # R3: same defect as the branch above, reached via cancel/abort instead of
1818
+ # pool exhaustion. Preserve the attempt's concrete failure when pool repair
1819
+ # failed first and installed a generic infrastructure abort.
1820
+ if pending.outcome.failure is not None:
1821
+ abort_holder[0] = pending.outcome.failure
1822
+ return await self._emit_pending_retry_receipt(scenario, pending)
1823
+ next_index, next_runtime = next_leased
1824
+ return await self._run_scenario(
1825
+ scenario,
1826
+ scenario_index,
1827
+ attempt=2,
1828
+ tried=tried | {world_index},
1829
+ pre_leased=(next_index, next_runtime),
1830
+ pending_retry=pending,
1831
+ abort_holder=abort_holder,
1832
+ context=context,
1833
+ )
1834
+
1835
+ # M13: a plain terminal receipt is either a real passed/failed verdict (release — the
1836
+ # world is fine) or a non-retryable fault from `_fault()`. For the latter, an
1837
+ # exception/overrun code means the world is half-applied and must be discarded rather
1838
+ # than handed to the next scenario; `ready_not_ready` is a clean verdict and keeps
1839
+ # `release()`.
1840
+ if (
1841
+ outcome.failure is not None
1842
+ and outcome.failure.code in _DISCARD_ON_ERROR_CODES
1843
+ ):
1844
+ await self._pool.mark_unhealthy(
1845
+ world_index, cause=outcome.failure.message
1846
+ )
1847
+ else:
1848
+ await self._pool.release(world_index)
1849
+ world_resolved = True
1850
+ await self._emit(self._outbound.receipt(outcome), what="receipt")
1851
+ return outcome
1852
+ finally:
1853
+ if not world_resolved:
1854
+ if self._pool.fenced is not None:
1855
+ # The exception that skipped every path above was a fence (or the pool was
1856
+ # already fenced by something else) -- the world itself never did anything
1857
+ # wrong. `mark_unhealthy()` here would emit a false `world_unhealthy` after the
1858
+ # run already stopped emitting, and schedule a `provision()` reconcile for a
1859
+ # job that is not coming back for it.
1860
+ await self._pool.release(world_index)
1861
+ else:
1862
+ # Something blew past every handled path above (a bug in this module
1863
+ # itself) — the world must not be silently stranded outside the pool's
1864
+ # bookkeeping. Discarded rather than released: an exception here leaves its
1865
+ # state unknown, and world-handle-interface.md's own exception rule is
1866
+ # "discarded and re-provisioned, never reused."
1867
+ await self._pool.mark_unhealthy(
1868
+ world_index,
1869
+ cause="scenario driver crashed while holding this world",
1870
+ )
1871
+
1872
+ async def _execute(
1873
+ self,
1874
+ scenario: Scenario,
1875
+ world: World,
1876
+ runtime: EnvironmentRuntime,
1877
+ world_index: int,
1878
+ *,
1879
+ attempt: int,
1880
+ context: "_ScenarioContext",
1881
+ ) -> "ResultReceipt | _Retry":
1882
+ setup = await _run_phase(
1883
+ scenario.setup,
1884
+ world,
1885
+ timeout=SETUP_TIMEOUT_SECONDS,
1886
+ phase="setup",
1887
+ executor=self._executor,
1888
+ )
1889
+ if setup.failure is not None:
1890
+ return self._fault(
1891
+ scenario,
1892
+ world_index,
1893
+ attempt,
1894
+ setup.failure,
1895
+ sub_goals=_unjudged(scenario.sub_goals),
1896
+ )
1897
+
1898
+ read_only = world.read_only()
1899
+ ready = await _run_phase(
1900
+ scenario.ready,
1901
+ read_only,
1902
+ timeout=READY_TIMEOUT_SECONDS,
1903
+ phase="ready",
1904
+ executor=self._executor,
1905
+ )
1906
+ if ready.failure is not None:
1907
+ return self._fault(
1908
+ scenario,
1909
+ world_index,
1910
+ attempt,
1911
+ ready.failure,
1912
+ sub_goals=_unjudged(scenario.sub_goals),
1913
+ )
1914
+ verdict = _classify_ready(ready.value)
1915
+ if verdict.broken:
1916
+ return self._fault(
1917
+ scenario,
1918
+ world_index,
1919
+ attempt,
1920
+ _failure("ready_broken", f"ready() returned {ready.value!r}"),
1921
+ sub_goals=_unjudged(scenario.sub_goals),
1922
+ )
1923
+ if not verdict.held:
1924
+ # The verdict says the precondition did not hold, never whether the setup that was
1925
+ # supposed to establish it ran. Without that, a scenario that never dials looks the
1926
+ # same whether its setup failed or its check is wrong.
1927
+ logger.warning(
1928
+ "scenario %s not ready on world %s: %s (setup reported: %s)",
1929
+ scenario.scenario_key,
1930
+ world_index,
1931
+ verdict.reason or "no reason given",
1932
+ getattr(setup, "value", None),
1933
+ )
1934
+ return self._fault(
1935
+ scenario,
1936
+ world_index,
1937
+ attempt,
1938
+ _failure("ready_not_ready", verdict.reason or ""),
1939
+ sub_goals=_unjudged(scenario.sub_goals),
1940
+ )
1941
+
1942
+ try:
1943
+ call_outcome = await _run_call(self._call_runner, scenario, runtime, world)
1944
+ except WorldUnavailable as exc:
1945
+ return _Retry(
1946
+ _failure("world_unavailable", str(exc)),
1947
+ sub_goals=_unjudged(scenario.sub_goals),
1948
+ call=None,
1949
+ mark_unhealthy=True,
1950
+ )
1951
+ except CallAborted as exc:
1952
+ call = self._call_summary(exc.partial)
1953
+ return self._fault(
1954
+ scenario,
1955
+ world_index,
1956
+ attempt,
1957
+ _failure(exc.code, str(exc)),
1958
+ sub_goals=_unjudged(scenario.sub_goals),
1959
+ call=call,
1960
+ )
1961
+ except Exception as exc: # noqa: BLE001
1962
+ # B3: the call runner crashing outright (not a `CallAborted` it chose to raise) is the
1963
+ # same world-handle-interface.md v3.3 row — "the simulated-call machinery crashed" —
1964
+ # just with no partial evidence to report.
1965
+ return self._fault(
1966
+ scenario,
1967
+ world_index,
1968
+ attempt,
1969
+ _failure("call_failed", f"{type(exc).__name__}: {exc}"),
1970
+ sub_goals=_unjudged(scenario.sub_goals),
1971
+ call=None,
1972
+ )
1973
+
1974
+ # Set the moment the call step returns, so a crash later in this method (e.g.
1975
+ # `world.read_only()` below) still reports the call that genuinely ran, not `null`.
1976
+ context.call = self._call_summary(call_outcome)
1977
+ calls = list(
1978
+ call_outcome.calls
1979
+ ) # m12: `folder.py::_RUNNABLE` expects a list, not a tuple.
1980
+ if not calls and getattr(scenario, "requires_tool_evidence", True):
1981
+ # M10: unconditioned on `turns` — an empty list must never reach checks regardless of
1982
+ # whether the simulator observed a turn (world-handle-interface.md "Coverage
1983
+ # guarantee": "An empty list is never handed to checks").
1984
+ failure = _failure(
1985
+ "evidence_missing",
1986
+ "no tool calls were captured for this scenario's call",
1987
+ )
1988
+ return _Retry(
1989
+ failure,
1990
+ sub_goals=_unjudged(scenario.sub_goals),
1991
+ call=self._call_summary(call_outcome),
1992
+ mark_unhealthy=False,
1993
+ )
1994
+
1995
+ if not scenario.sub_goals:
1996
+ # m7: `all(())` is vacuously True — a scenario declaring zero sub-goals must not read
1997
+ # as a silent pass.
1998
+ return self._fault(
1999
+ scenario,
2000
+ world_index,
2001
+ attempt,
2002
+ _failure(
2003
+ "check_broken",
2004
+ "scenario declared zero sub_goals — a vacuous pass is forbidden",
2005
+ ),
2006
+ sub_goals=(),
2007
+ call=self._call_summary(call_outcome),
2008
+ )
2009
+
2010
+ sub_goal_results: list[SubGoalResult] = []
2011
+ judged_pending: list[tuple[int, Any]] = []
2012
+ check_handle = world.read_only()
2013
+ broken_failure: ReceiptFailure | None = None
2014
+ for goal in scenario.sub_goals:
2015
+ if broken_failure is not None:
2016
+ sub_goal_results.append(
2017
+ SubGoalResult(
2018
+ name=goal.name, held=None, reason=None, judged=goal.judged != ""
2019
+ )
2020
+ )
2021
+ continue
2022
+ outcome = await _run_phase(
2023
+ goal.check,
2024
+ check_handle,
2025
+ calls,
2026
+ timeout=CHECK_TIMEOUT_SECONDS,
2027
+ phase="check",
2028
+ executor=self._executor,
2029
+ )
2030
+ if outcome.failure is not None:
2031
+ if outcome.failure.code == "world_unavailable":
2032
+ return _Retry(
2033
+ outcome.failure,
2034
+ sub_goals=tuple(sub_goal_results)
2035
+ + _unjudged([goal])
2036
+ + _unjudged(scenario.sub_goals[len(sub_goal_results) + 1 :]),
2037
+ call=self._call_summary(call_outcome),
2038
+ mark_unhealthy=True,
2039
+ )
2040
+ broken_failure = outcome.failure
2041
+ sub_goal_results.append(
2042
+ SubGoalResult(
2043
+ name=goal.name, held=None, reason=None, judged=goal.judged != ""
2044
+ )
2045
+ )
2046
+ continue
2047
+ if goal.judged:
2048
+ # A judged sub-goal has no code to settle it: a model decides, here, while the
2049
+ # world the call left behind is still alive. Collected and run together below.
2050
+ judged_pending.append((len(sub_goal_results), goal))
2051
+ sub_goal_results.append(
2052
+ SubGoalResult(name=goal.name, held=None, reason=None, judged=True)
2053
+ )
2054
+ continue
2055
+ verdict = _classify_check(outcome.value)
2056
+ if verdict.broken:
2057
+ broken_failure = _failure(
2058
+ "check_broken", f"{goal.name}: check() returned {outcome.value!r}"
2059
+ )
2060
+ sub_goal_results.append(
2061
+ SubGoalResult(
2062
+ name=goal.name, held=None, reason=None, judged=goal.judged != ""
2063
+ )
2064
+ )
2065
+ continue
2066
+ sub_goal_results.append(
2067
+ SubGoalResult(
2068
+ name=goal.name,
2069
+ held=verdict.held,
2070
+ reason=_sub_goal_reason(goal, verdict),
2071
+ judged=goal.judged != "",
2072
+ )
2073
+ )
2074
+
2075
+ if judged_pending:
2076
+ # Judged sub-goals only read, so they are independent of each other and of the coded
2077
+ # checks: one round trip for all of them rather than one each.
2078
+ async def _settle(goal: Any) -> Any:
2079
+ # Awaited, not called inline: calling an injected judge whose signature does not
2080
+ # match raises while the coroutines are still being built, which is outside
2081
+ # `gather`'s net and errors the scenario. Inside a coroutine it is just a fault.
2082
+ return await self._judge(
2083
+ goal, check_handle, calls, messages=call_outcome.messages
2084
+ )
2085
+
2086
+ verdicts = await asyncio.gather(
2087
+ *(_settle(goal) for _, goal in judged_pending),
2088
+ return_exceptions=True,
2089
+ )
2090
+ for (slot, goal), outcome in zip(judged_pending, verdicts):
2091
+ if isinstance(outcome, BaseException):
2092
+ held, why = None, f"the judge could not run: {outcome!r}"
2093
+ else:
2094
+ try:
2095
+ held, why = outcome
2096
+ except (TypeError, ValueError):
2097
+ # An injected judge that answers in some other shape is unreadable, not
2098
+ # authoritative. Unpacking it here would raise inside `_grade` and error
2099
+ # the whole scenario, which is the one thing a verdict must never do.
2100
+ held, why = (
2101
+ None,
2102
+ f"the judge returned no usable verdict: {outcome!r}",
2103
+ )
2104
+ sub_goal_results[slot] = SubGoalResult(
2105
+ name=goal.name, held=held, reason=why, judged=True
2106
+ )
2107
+
2108
+ if broken_failure is not None:
2109
+ return self._fault(
2110
+ scenario,
2111
+ world_index,
2112
+ attempt,
2113
+ broken_failure,
2114
+ sub_goals=tuple(sub_goal_results),
2115
+ call=self._call_summary(call_outcome),
2116
+ )
2117
+
2118
+ # A sub-goal the judge did not settle is reported unsettled on the sub-goal itself and
2119
+ # never decides the scenario: the call ran, its evidence stands, and a model that could
2120
+ # not answer is a fault of neither the agent nor the run. Only a settled `False` fails a
2121
+ # scenario. `errored` stays reachable for a call or infrastructure fault, which is raised
2122
+ # elsewhere; nothing about a verdict produces one.
2123
+ if any(result.held is False for result in sub_goal_results):
2124
+ status = "failed"
2125
+ else:
2126
+ status = "passed"
2127
+ failure = None
2128
+ return ResultReceipt(
2129
+ scenario_key=scenario.scenario_key,
2130
+ scenario_id=scenario.scenario_id,
2131
+ scenario_attempt=attempt,
2132
+ world_index=world_index,
2133
+ status=status,
2134
+ sub_goals=tuple(sub_goal_results),
2135
+ evaluations=(),
2136
+ call=self._call_summary(call_outcome),
2137
+ failure=failure,
2138
+ )
2139
+
2140
+ @staticmethod
2141
+ def _call_summary(outcome: CallOutcome | None) -> CallSummary | None:
2142
+ if outcome is None:
2143
+ return None
2144
+ return CallSummary(
2145
+ started_at=outcome.started_at,
2146
+ ended_at=outcome.ended_at,
2147
+ duration_ms=outcome.duration_ms,
2148
+ turns=outcome.turns,
2149
+ transcript_artifact=outcome.transcript_artifact,
2150
+ recording_artifacts=outcome.recording_artifacts,
2151
+ stop_reason=outcome.stop_reason,
2152
+ )
2153
+
2154
+ def _fault(
2155
+ self,
2156
+ scenario: Scenario,
2157
+ world_index: int,
2158
+ attempt: int,
2159
+ failure: ReceiptFailure,
2160
+ *,
2161
+ sub_goals: tuple[SubGoalResult, ...],
2162
+ call: CallSummary | None = None,
2163
+ retry: bool | None = None,
2164
+ ) -> "ResultReceipt | _Retry":
2165
+ should_retry = _is_retryable(failure.code) if retry is None else retry
2166
+ if should_retry:
2167
+ return _Retry(
2168
+ failure,
2169
+ sub_goals=sub_goals,
2170
+ call=call,
2171
+ mark_unhealthy=failure.code == "world_unavailable",
2172
+ )
2173
+ return ResultReceipt(
2174
+ scenario_key=scenario.scenario_key,
2175
+ scenario_id=scenario.scenario_id,
2176
+ scenario_attempt=attempt,
2177
+ world_index=world_index,
2178
+ status="errored",
2179
+ sub_goals=sub_goals,
2180
+ evaluations=(),
2181
+ call=call,
2182
+ failure=failure,
2183
+ )
2184
+
2185
+
2186
+ @dataclass(frozen=True)
2187
+ class _Retry:
2188
+ failure: ReceiptFailure
2189
+ sub_goals: tuple[SubGoalResult, ...]
2190
+ call: CallSummary | None
2191
+ mark_unhealthy: bool
2192
+
2193
+
2194
+ __all__ = [
2195
+ "CHECK_TIMEOUT_SECONDS",
2196
+ "READY_TIMEOUT_SECONDS",
2197
+ "SETUP_TIMEOUT_SECONDS",
2198
+ "Call",
2199
+ "CallAborted",
2200
+ "CallOutcome",
2201
+ "CallRunner",
2202
+ "CallSummary",
2203
+ "Evaluation",
2204
+ "HostedScheduler",
2205
+ "NoWorldsAvailable",
2206
+ "OutboundPort",
2207
+ "ReadOnlyWorld",
2208
+ "ReceiptFailure",
2209
+ "ResultReceipt",
2210
+ "RunResult",
2211
+ "Scenario",
2212
+ "SubGoal",
2213
+ "SubGoalResult",
2214
+ "World",
2215
+ "WorldFactory",
2216
+ "WorldPool",
2217
+ "WorldProvisioner",
2218
+ ]