agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,2402 @@
1
+ """The hosted guest's `main()` — `hosted-execution-seams.md` v1.14 §0/§4/§5, `outbound-channels.md`
2
+ v1.3, `world-handle-interface.md` v3.4. Everything between "sandbox starts" and "exit code": read
3
+ `/work/job.json`, load the platform capability file, run §2e preflight, pre-allocate scenarios
4
+ against `endpoints.scenarios`, provision the world pool, drive the scenario loop, adapt its events/
5
+ receipts/artifacts onto the real outbound clients, and honor the exit-code contract (§0.6).
6
+
7
+ Ownership boundary (read this before touching orchestration order): the stages BEFORE bundle
8
+ authoring — `understanding_agent`, `generating_environment`, `building_environment`'s bundle-write
9
+ half — belong to Rishav's stages (contract §6) and are not implemented anywhere in this repo yet.
10
+ This module does not attempt them. `BundleSource`/`ScenarioSource` below are the seams a later
11
+ change wires the real stages through; until then their defaults raise a typed, clearly-named error
12
+ rather than silently producing a fake bundle or a fake scenario set.
13
+
14
+ `process_runtime.py`, `hosted_scheduler.py`, and `outbound.py` were being fixed by parallel workers
15
+ while this module was written. It codes against the four frozen contracts and the cross-review
16
+ obligation lists, not against those files' exact HEAD.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import argparse
22
+ import asyncio
23
+ import hashlib
24
+ import json
25
+ import logging
26
+ import os
27
+ import random
28
+ import signal
29
+ import time
30
+ import uuid
31
+ from dataclasses import dataclass, field
32
+ from datetime import datetime, timezone
33
+ from pathlib import Path
34
+ from typing import Any, Callable, Protocol, Sequence
35
+
36
+ from . import observability
37
+ from . import outbound as ob
38
+ from .bundle_v2 import BundleV2Error, EnvironmentBundleV2, load_bundle_v2
39
+ from .call_runner import CallRunnerContext, CallRunnerImpl
40
+ from .hosted_scheduler import (
41
+ CallOutcome,
42
+ CallRunner,
43
+ HostedScheduler,
44
+ ResultReceipt,
45
+ RunResult,
46
+ Scenario,
47
+ World,
48
+ WorldFactory,
49
+ WorldPool,
50
+ WorldProvisioner,
51
+ )
52
+ from .job import (
53
+ ArtifactLevel,
54
+ ExecutionMode,
55
+ FailureDomain,
56
+ HarnessArtifactPolicy,
57
+ HarnessJob,
58
+ HarnessStage,
59
+ )
60
+ from .process_preflight import PreflightError, preflight_bundle
61
+ from .process_runtime import (
62
+ SECTION_2F_DOMAIN,
63
+ EnvironmentRuntime,
64
+ ProcessRuntimeError,
65
+ ProcessRuntimeProvider,
66
+ RuntimeEndpoint,
67
+ )
68
+ from .scenario_source import (
69
+ BundleScenarioSource,
70
+ ScenarioDocumentInvalid,
71
+ bundle_has_scenarios,
72
+ )
73
+ from .world.handle import HostedWorld
74
+ from .world.stores.postgres import AttachedPostgresStore
75
+
76
+ logger = logging.getLogger(__name__)
77
+
78
+
79
+ class _JobIdFilter(logging.Filter):
80
+ """Stamp every log record with the job this runner is serving.
81
+
82
+ One runner process serves exactly one job, so the id is process-wide rather than per-task
83
+ state. Concurrent runs are separate processes, but their stdout is collected into one place,
84
+ and a line with no job id cannot be attributed to a run at all -- which is the difference
85
+ between reading a log and guessing at it.
86
+ """
87
+
88
+ def __init__(self) -> None:
89
+ super().__init__()
90
+ self.job_id = "-"
91
+
92
+ def filter(self, record: logging.LogRecord) -> bool:
93
+ if not hasattr(record, "job_id"):
94
+ record.job_id = self.job_id
95
+ return True
96
+
97
+
98
+ _JOB_ID_FILTER = _JobIdFilter()
99
+
100
+
101
+ def configure_runner_logging(job_id: str | None) -> None:
102
+ """Put the job id on every line this process emits, including libraries' lines.
103
+
104
+ Installed on the root logger rather than ours, because the lines that are hardest to attribute
105
+ are the ones from livekit, httpx and the model clients.
106
+ """
107
+ _JOB_ID_FILTER.job_id = str(job_id or "-")
108
+ root = logging.getLogger()
109
+ for handler in root.handlers:
110
+ handler.addFilter(_JOB_ID_FILTER)
111
+ if not root.handlers:
112
+ handler = logging.StreamHandler()
113
+ handler.addFilter(_JOB_ID_FILTER)
114
+ root.addHandler(handler)
115
+ root.setLevel(logging.INFO)
116
+ for handler in logging.getLogger().handlers:
117
+ handler.setFormatter(
118
+ logging.Formatter(
119
+ "%(asctime)s %(levelname)s job=%(job_id)s %(name)s: %(message)s"
120
+ )
121
+ )
122
+
123
+ # --- §0.6 exit-code contract --------------------------------------------------------------------
124
+ #
125
+ # 0 = any terminal stage reached (completed/failed/canceled), outbox flushed. 3 = fenced/superseded
126
+ # (HostedFencedError anywhere -> stop emitting, no terminal event, exit 3). 4 = the terminal was
127
+ # decided but the final drain could not deliver it (the events channel failed, or the platform
128
+ # permanently rejected the terminal item itself) -- the gateway treats it exactly like a crash
129
+ # (infrastructure retry, fresh channels), but the distinct code tells operators the job DID reach a
130
+ # terminal state, unlike a genuine crash. Any other non-zero = the guest crashed before a terminal
131
+ # state -- the gateway records `infrastructure`. Capabilities-file failures are explicitly carved
132
+ # out of the "any other non-zero" bucket only by CODE (they must never be 3, per
133
+ # outbound-channels.md v1.3's rejection table); they still use a non-zero exit here since there is
134
+ # no channel to report a terminal FAILED event through.
135
+ EXIT_OK = 0
136
+ EXIT_FENCED = 3
137
+ EXIT_TERMINAL_UNDELIVERED = (
138
+ 4 # terminal reached but not provably flushed on the final drain.
139
+ )
140
+ EXIT_BOOT_FAILURE = (
141
+ 1 # capabilities.json could not be loaded -- no channel, no event (v1.3 table).
142
+ )
143
+ EXIT_CRASHED = 2 # an uncaught failure before any terminal stage was reached.
144
+
145
+ # Cancellation signal (spine §0 step 7 / outbound-channels.md "Cancellation signal"). The task
146
+ # brief that spawned this module named `/work/cancel.json`; the two frozen contracts that actually
147
+ # define this file (seams §0 step 7, outbound-channels "Cancellation signal") both name
148
+ # `/run/futureagi/cancel.json`. Contracts are authoritative over a task brief.
149
+ CANCEL_SIGNAL_PATH = "/run/futureagi/cancel.json"
150
+
151
+ # STUCK DECISION (fail-safe/reversible; contract gap): the invocation contract
152
+ # (spine §0 step 5) pins the entrypoint's argv to exactly `job --source ... --output ...`; `--output`
153
+ # is `/work/artifacts` (spine layout block), so `work_directory` (what `preflight_bundle`/
154
+ # `provision`/`write_build_output` all want -- the `/work` root) is derived as `output.parent`
155
+ # rather than taken as a separate flag, since the frozen invocation line has no room for one.
156
+ # `bundle_dir` has no convention anywhere in the frozen documents at all (bundle authoring is not
157
+ # built yet); `DEFAULT_BUNDLE_DIR_NAME` is this module's own placeholder location, overridable via
158
+ # `BundleSource` injection so a later change can point it at wherever the real authoring stage ends
159
+ # up writing without touching this file's orchestration.
160
+ DEFAULT_BUNDLE_DIR_NAME = "bundle"
161
+ EVENTS_SPOOL_DIR_NAME = (
162
+ "outbound-spool" # must not live under work_directory/"artifacts".
163
+ )
164
+
165
+ SECRETS_PATH = Path("/run/futureagi/secrets.json")
166
+ SIMULATOR_SECRETS_PATH = Path("/run/futureagi/simulator-secrets.json")
167
+
168
+ # Platform-owned simulator configuration is delivered on a separate control-plane channel. It
169
+ # must never be confused with the customer's ``target_provider`` refs, which are selectively
170
+ # injected into the untrusted agent processes by ProcessRuntimeProvider. These names are the
171
+ # complete set the in-process text/voice simulators may consume.
172
+ _SIMULATOR_SECRET_ALIASES = frozenset(
173
+ {
174
+ "ALK_BACKGROUND_NOISE",
175
+ "ALK_BACKGROUND_NOISE_CATALOG",
176
+ "ALK_HARNESS",
177
+ "ALK_VOICEMAIL_SCENARIOS",
178
+ "ALK_HARNESS_MODEL",
179
+ "ALK_HARNESS_THINKING",
180
+ "ALK_VERTEX_LOCATION",
181
+ "CARTESIA_API_KEY",
182
+ "DEEPGRAM_API_KEY",
183
+ # Observe configuration: the platform's own account, never the customer's.
184
+ "FI_API_KEY",
185
+ "FI_BASE_URL",
186
+ "FI_HARNESS_PROJECT",
187
+ "FI_SECRET_KEY",
188
+ "GEMINI_API_KEY",
189
+ "GOOGLE_API_KEY",
190
+ "GOOGLE_APPLICATION_CREDENTIALS",
191
+ "GOOGLE_CLOUD_LOCATION",
192
+ "GOOGLE_CLOUD_PROJECT",
193
+ "GOOGLE_GENAI_USE_VERTEXAI",
194
+ "HARNESS_BACKGROUND_NOISE_VOLUME",
195
+ "HARNESS_OBSERVABILITY",
196
+ "LIVEKIT_URL",
197
+ "LIVEKIT_API_KEY",
198
+ "LIVEKIT_API_SECRET",
199
+ "OPENAI_API_KEY",
200
+ "SIMULATOR_LLM_MODEL",
201
+ "SIMULATOR_LLM_PROVIDER",
202
+ "SIMULATOR_STT_MODEL",
203
+ "SIMULATOR_STT_PROVIDER",
204
+ "SIMULATOR_TTS_MODEL",
205
+ "SIMULATOR_TTS_PROVIDER",
206
+ }
207
+ )
208
+
209
+
210
+ # =================================================================================================
211
+ # Boot -- job.json + capabilities.json (§0.2/§0.4; outbound-channels.md Authentication).
212
+ # =================================================================================================
213
+
214
+
215
+ def load_job(job_path: Path) -> HarnessJob:
216
+ """§0.2: `/work/job.json` is the provisioner's job-identity and configuration source."""
217
+ job = HarnessJob.model_validate_json(job_path.read_text(encoding="utf-8"))
218
+ if job.execution is not ExecutionMode.HOSTED:
219
+ raise ValueError("hosted_entrypoint_requires_hosted_job")
220
+ return job
221
+
222
+
223
+ def resolve_parallelism(job: HarnessJob) -> int:
224
+ """`job.runtime.parallelism` = W (glossary). Returns the RAW requested value, never
225
+ clamped -- §2e.7 reserves `parallelism_out_of_range` for a W outside 1..8, and
226
+ `preflight_bundle` (called BEFORE any provisioning) is the enforcement point for the UPPER
227
+ bound. The lower bound never reaches preflight at all: `RuntimeRequirements.parallelism`'s own
228
+ `ge=1` rejects a non-positive W earlier, at `load_job`, as a deliberate defense-in-depth floor
229
+ (harmless today since the gateway caps W at admission before a job is ever built). Clamping
230
+ here would silently launder an in-range-but-wrong W and make `parallelism_out_of_range`
231
+ permanently unreachable for the upper bound."""
232
+ return job.runtime.parallelism
233
+
234
+
235
+ def job_secret_purposes(job: HarnessJob) -> dict[str, str]:
236
+ """§1: `agent.secret_refs` alias -> `SecretRef.purpose`, the shape `preflight_bundle` wants."""
237
+ return {alias: ref.purpose for alias, ref in job.agent.secret_refs.items()}
238
+
239
+
240
+ def peek_secret_values(secrets_path: Path) -> tuple[str, ...]:
241
+ """A non-destructive read of `/run/futureagi/secrets.json`'s VALUES ONLY, for outbound
242
+ redaction (`extra_secret_values` — outbound.py's `redact_outbound_text`). §0.3's lifetime rule
243
+ ("the provisioner loads this file into memory at startup and deletes it") is honored by
244
+ `ProcessRuntimeProvider` itself; this is an additional, side-effect-free read (no unlink) done
245
+ once at boot so free-text event/log/failure fields can be scrubbed of every resolved secret
246
+ value, not just URL userinfo. Never fatal: a missing/malformed file just means no extra values
247
+ to scrub, matching `redact_outbound_text`'s own `extra_secret_values=()` default."""
248
+ try:
249
+ raw = json.loads(secrets_path.read_text(encoding="utf-8"))
250
+ except (OSError, ValueError):
251
+ return ()
252
+ if not isinstance(raw, dict):
253
+ return ()
254
+ return tuple(str(value) for value in raw.values() if value)
255
+
256
+
257
+ def peek_secret_values_for_purpose(
258
+ secrets_path: Path,
259
+ secret_purposes: dict[str, str],
260
+ purpose: str,
261
+ ) -> dict[str, str]:
262
+ """The same non-destructive, no-unlink read as `peek_secret_values` (same file, same timing
263
+ constraint -- called BEFORE `pool.start()`, which is what actually deletes the file), but
264
+ ALIAS-preserving and filtered to one explicit purpose -- `peek_secret_values` throws the
265
+ alias away, which is fine for outbound redaction (it only needs the raw values) but useless for
266
+ the real `CallRunner`, which needs to pick e.g. `LIVEKIT_API_KEY` out of the map by name. Never
267
+ fatal: a missing/malformed file just means no target-provider secrets are available yet,
268
+ matching `CallRunnerImpl`'s own pre-dial validation (it reports the gap as a typed
269
+ `CallAborted`, never crashes on an empty map)."""
270
+ try:
271
+ raw = json.loads(secrets_path.read_text(encoding="utf-8"))
272
+ except (OSError, ValueError):
273
+ return {}
274
+ if not isinstance(raw, dict):
275
+ return {}
276
+ return {
277
+ str(alias): str(value)
278
+ for alias, value in raw.items()
279
+ if secret_purposes.get(str(alias)) == purpose
280
+ }
281
+
282
+
283
+ def peek_target_provider_secret_values(
284
+ secrets_path: Path, secret_purposes: dict[str, str]
285
+ ) -> dict[str, str]:
286
+ return peek_secret_values_for_purpose(
287
+ secrets_path, secret_purposes, "target_provider"
288
+ )
289
+
290
+
291
+ def peek_simulator_provider_secret_values(
292
+ secrets_path: Path, secret_purposes: dict[str, str]
293
+ ) -> dict[str, str]:
294
+ return peek_secret_values_for_purpose(
295
+ secrets_path, secret_purposes, "simulator_provider"
296
+ )
297
+
298
+
299
+ def load_simulator_secret_values(path: Path) -> dict[str, str]:
300
+ """Load and immediately remove the platform-owned simulator secret channel.
301
+
302
+ The fixed allowlist is deliberate: a platform deployment cannot accidentally use this file
303
+ to inject arbitrary ambient variables into the control process. Agent subprocesses still do
304
+ not inherit these values because ``process_runtime`` starts them from its closed environment
305
+ allowlist plus purpose-matched target secrets.
306
+ """
307
+ try:
308
+ raw = json.loads(path.read_text(encoding="utf-8"))
309
+ except (OSError, ValueError):
310
+ return {}
311
+ finally:
312
+ try:
313
+ path.unlink(missing_ok=True)
314
+ except OSError as exc:
315
+ logger.warning("simulator-secrets.json unlink failed: %s", exc)
316
+ if not isinstance(raw, dict):
317
+ return {}
318
+ return {
319
+ str(alias): str(value)
320
+ for alias, value in raw.items()
321
+ if str(alias) in _SIMULATOR_SECRET_ALIASES and value not in (None, "")
322
+ }
323
+
324
+
325
+ # =================================================================================================
326
+ # Bundle source -- §2 bundle authoring is not this module's (or built anywhere yet); injectable.
327
+ # =================================================================================================
328
+
329
+
330
+ class BundleUnavailableError(RuntimeError):
331
+ """Raised by a `BundleSource` when no bundle could be produced/located. Mapped the same way as
332
+ a `PreflightError` (FAILED, `FailureDomain.ENVIRONMENT`, stage `validating_environment`) —
333
+ from the entrypoint's point of view "no bundle" and "bad bundle" are the same class of
334
+ environment-authoring fault, and §2e's own failure table has no separate code for it."""
335
+
336
+ def __init__(self, code: str, message: str) -> None:
337
+ self.code = code
338
+ self.message = message
339
+ super().__init__(f"{code}: {message}")
340
+
341
+
342
+ class BundleSource(Protocol):
343
+ def load(
344
+ self, job: HarnessJob, *, source: Path, work_directory: Path
345
+ ) -> tuple[EnvironmentBundleV2, Path]: ...
346
+
347
+
348
+ # §2e's closed failure-code table (hosted-execution-seams.md) -- `BundleV2Error` has no typed
349
+ # `.code` (a bare `RuntimeError`), so `DefaultBundleSource.load` below string-splits
350
+ # its message on ":". `bundle_manifest_missing` (one of the four messages `load_bundle_v2` can
351
+ # raise) is not in this table -- a real contract gap -- so both that code AND anything else the
352
+ # split produces outside this frozen set fall back to `bundle_manifest_invalid` rather than
353
+ # shipping an unlisted code across the outbound seam.
354
+ _SECTION_2E_CODES = frozenset(
355
+ {
356
+ "compose_not_hosted",
357
+ "engine_unsupported",
358
+ "no_sql_store",
359
+ "seed_missing",
360
+ "seed_strategy_unsupported",
361
+ "sentinel_shape_mismatch",
362
+ "store_protocol_unsupported",
363
+ "capability_engine_mismatch",
364
+ "store_service_not_managed",
365
+ "reserved_name",
366
+ "unknown_placeholder",
367
+ "unknown_field",
368
+ "secret_in_bundle",
369
+ "secret_unclaimed",
370
+ "secret_missing",
371
+ "secret_purpose_forbidden",
372
+ "build_requires_root",
373
+ "user_assignment_invalid",
374
+ "configuration_name_duplicate",
375
+ "configuration_name_required",
376
+ "configuration_name_reserved",
377
+ "sentinel_shape_invalid",
378
+ "capability_unresolved",
379
+ "service_unresolved",
380
+ "control_service_unresolved",
381
+ "process_name_duplicate",
382
+ "inputs_digest_mismatch",
383
+ "bundle_schema_unsupported",
384
+ "bundle_manifest_invalid",
385
+ "bundle_manifest_drifted",
386
+ "bundle_digest_mismatch",
387
+ "bundle_digest_invalid",
388
+ "inputs_digest_invalid",
389
+ "file_sha256_invalid",
390
+ "source_digest_invalid",
391
+ "bundle_file_missing",
392
+ "bundle_file_changed",
393
+ "bundle_file_unlisted",
394
+ "bundle_symlink_forbidden",
395
+ "bundle_path_unsafe",
396
+ "depends_on_unresolved",
397
+ "depends_on_cycle",
398
+ "seed_file_missing",
399
+ "seed_file_unlisted",
400
+ "process_count_exceeded",
401
+ "parallelism_out_of_range",
402
+ "evidence_seam_required",
403
+ "processes_required",
404
+ "processes_and_seed_forbidden",
405
+ "document_only_for_compose",
406
+ "compose_runtime_requires_document",
407
+ "build_command_step_empty",
408
+ "started_check_requires_exactly_one_of_port_or_log_marker",
409
+ "resolved_secret_forbidden",
410
+ "capability_slug_invalid",
411
+ "process_name_invalid",
412
+ "fixed_port_reserved",
413
+ }
414
+ )
415
+
416
+
417
+ def _bundle_unavailable_code(raw_message: str) -> str:
418
+ code = raw_message.split(":", 1)[0].strip()
419
+ return code if code in _SECTION_2E_CODES else "bundle_manifest_invalid"
420
+
421
+
422
+ class DefaultBundleSource:
423
+ """Looks for an already-authored bundle at `work_directory / bundle_dir_name`. This is a
424
+ placeholder location this module invented (see the module-level STUCK DECISION note) — a real
425
+ bundle-authoring stage should either write there or be wired in via its own `BundleSource`."""
426
+
427
+ def __init__(self, bundle_dir_name: str = DEFAULT_BUNDLE_DIR_NAME) -> None:
428
+ self._bundle_dir_name = bundle_dir_name
429
+
430
+ def load(
431
+ self, job: HarnessJob, *, source: Path, work_directory: Path
432
+ ) -> tuple[EnvironmentBundleV2, Path]:
433
+ del job, source # unused by the default (a real stage would author from these)
434
+ bundle_dir = work_directory / self._bundle_dir_name
435
+ try:
436
+ manifest = load_bundle_v2(bundle_dir)
437
+ except BundleV2Error as exc:
438
+ raise BundleUnavailableError(
439
+ _bundle_unavailable_code(exc.args[0]), str(exc)
440
+ ) from exc
441
+ return manifest, bundle_dir
442
+
443
+
444
+ # =================================================================================================
445
+ # Scenario source -- generation is a separate contract (in review, not available here); the
446
+ # pre-allocation CALL is this module's (ScenariosClient below). Injectable for the same reason as
447
+ # BundleSource: the glue between "generated scenarios" and "pre-allocated against the platform" can
448
+ # only be finished once that contract's payload shape lands.
449
+ # =================================================================================================
450
+
451
+
452
+ class ScenarioSourceNotWired(RuntimeError):
453
+ """The default `ScenarioSource` — no Scenario Generation Contract implementation exists in this
454
+ repo yet. Raised rather than fabricating scenarios, and mapped to FAILED / `platform_sync` /
455
+ `validating_scenarios`, matching spine §5 step 3.5's own failure mapping for a pre-allocation
456
+ that never completes."""
457
+
458
+
459
+ class ScenarioSource(Protocol):
460
+ async def build(
461
+ self,
462
+ job: HarnessJob,
463
+ bundle: EnvironmentBundleV2,
464
+ scenarios_client: "ScenariosClient",
465
+ *,
466
+ pool: WorldPool,
467
+ world_factory: WorldFactory,
468
+ bundle_dir: Path,
469
+ ) -> Sequence[Scenario]: ...
470
+
471
+
472
+ class NotWiredScenarioSource:
473
+ async def build(
474
+ self,
475
+ job: HarnessJob,
476
+ bundle: EnvironmentBundleV2,
477
+ scenarios_client: "ScenariosClient",
478
+ *,
479
+ pool: WorldPool,
480
+ world_factory: WorldFactory,
481
+ bundle_dir: Path,
482
+ ) -> Sequence[Scenario]:
483
+ del job, bundle, scenarios_client, pool, world_factory, bundle_dir
484
+ raise ScenarioSourceNotWired(
485
+ "no ScenarioSource wired -- scenario generation is not implemented in this repo yet "
486
+ "(Scenario Generation Contract, in review)"
487
+ )
488
+
489
+
490
+ # =================================================================================================
491
+ # WorldFactory -- real HostedWorld instances, fed by build.json's row counts (never
492
+ # a partial map).
493
+ # =================================================================================================
494
+
495
+
496
+ class WorldFactoryError(RuntimeError):
497
+ """The provisioner handed back a runtime this factory cannot build a `World` for — a bug
498
+ upstream (no postgres endpoint despite §2e's `no_sql_store` guarantee, or `build.json` missing
499
+ the row counts for that store), never a scenario-code fault."""
500
+
501
+
502
+ def _process_runtime_error_domain(exc: ProcessRuntimeError) -> FailureDomain:
503
+ """v1.15 §2f: the producer (`process_runtime.py`) resolves and carries `domain` at the raise
504
+ site -- read it directly rather than re-deriving `spawn_failed`'s managed/source split from
505
+ the manifest (the old approach could not tell which process kind failed without one). The
506
+ imported `SECTION_2F_DOMAIN` map is a fallback ONLY, for an error that reaches here with no
507
+ carried domain -- logged when it fires, matching the scheduler's own rule.
508
+ """
509
+ if exc.domain is not None:
510
+ return exc.domain
511
+ if exc.code in SECTION_2F_DOMAIN:
512
+ logger.warning(
513
+ "process_runtime error %r crossed the §4 seam with no carried domain; using the §2f "
514
+ "fallback map (%s)",
515
+ exc.code,
516
+ SECTION_2F_DOMAIN[exc.code].value,
517
+ )
518
+ return SECTION_2F_DOMAIN[exc.code]
519
+ return FailureDomain.INFRASTRUCTURE # internal_* etc. -- the honest default
520
+
521
+
522
+ _SECTION_2F_CODES: frozenset[str] = frozenset(SECTION_2F_DOMAIN)
523
+
524
+
525
+ def _section_2f_code(code: str) -> str:
526
+ # §2f is closed (contract §4.6) -- `process_runtime.py`'s own `internal_*` codes, and this
527
+ # module's untyped-exception fallback, must never cross the outbound seam unlabeled, matching
528
+ # the discipline `_bundle_unavailable_code` already applies to §2e. The real code is
529
+ # still visible on the wire -- it stays in `message` (`ProcessRuntimeError.__str__` embeds it,
530
+ # and the untyped-exception call site prefixes it explicitly).
531
+ return code if code in _SECTION_2F_CODES else "spawn_failed"
532
+
533
+
534
+ def _find_postgres_endpoint(runtime: EnvironmentRuntime) -> RuntimeEndpoint:
535
+ for endpoint in runtime.endpoints.values():
536
+ if endpoint.protocol == "postgres":
537
+ return endpoint
538
+ raise WorldFactoryError(
539
+ f"world {runtime.world_index}: no postgres-protocol endpoint in {sorted(runtime.endpoints)} "
540
+ "-- §2e's no_sql_store rule should make this unreachable"
541
+ )
542
+
543
+
544
+ def load_build_output(work_directory: Path) -> dict[str, Any]:
545
+ """`write_build_output` (process_runtime.py) writes `<work_directory>/artifacts/build.json`.
546
+ Read fresh each call — cheap, and the row counts are immutable after baseline freeze, so
547
+ re-reading is simpler than a cache invalidation story for the same modest cost."""
548
+ path = work_directory / "artifacts" / "build.json"
549
+ try:
550
+ return json.loads(path.read_text(encoding="utf-8"))
551
+ except (OSError, ValueError) as exc:
552
+ raise WorldFactoryError(f"build.json unreadable at {path}: {exc}") from exc
553
+
554
+
555
+ def row_counts_for_capability(
556
+ build_output: dict[str, Any], capability: str
557
+ ) -> dict[str, int]:
558
+ for store in build_output.get("stores", []):
559
+ if store.get("capability") == capability:
560
+ counts = store.get("row_counts") or {}
561
+ return {str(name): int(count) for name, count in counts.items()}
562
+ raise WorldFactoryError(
563
+ f"build.json has no store entry for capability {capability!r} — the provisioner "
564
+ "guarantees a complete row-count map per store, so this bundle's build output is malformed"
565
+ )
566
+
567
+
568
+ class ProcessWorldFactory:
569
+ """Builds a real `HostedWorld` over the runtime's postgres endpoint. `AttachedPostgresStore`
570
+ (not the bare `PostgresStore`) is the correct base here — it takes a raw DSN and never manages
571
+ a container's own lifecycle, matching a hosted world where `ProcessRuntimeProvider` already
572
+ owns the postgres process."""
573
+
574
+ def __init__(self, work_directory: Path) -> None:
575
+ self._work_directory = work_directory
576
+
577
+ async def create(self, runtime: EnvironmentRuntime, *, rng: random.Random) -> World:
578
+ endpoint = _find_postgres_endpoint(runtime)
579
+ build_output = await asyncio.to_thread(load_build_output, self._work_directory)
580
+ row_counts = row_counts_for_capability(build_output, endpoint.capability)
581
+ store = AttachedPostgresStore(endpoint.address)
582
+ return await asyncio.to_thread(
583
+ HostedWorld, store, runtime.world_index, rng, row_counts
584
+ )
585
+
586
+
587
+ # =================================================================================================
588
+ # CallRunner -- the real voice track. Explicit LiveKit jobs and auto-discovered voice contracts
589
+ # use it. Explicit Vapi/Retell jobs remain outside the repository-hosted runner.
590
+ # =================================================================================================
591
+
592
+
593
+ class CallRunnerNotWired(RuntimeError):
594
+ """Raised by `NotWiredCallRunner`. `hosted_scheduler._execute` treats any exception out of
595
+ `CallRunner.run` (other than `WorldUnavailable`/`CallAborted`) as `call_failed`
596
+ (`FailureDomain.INFRASTRUCTURE`, retried once) — so a job run with nothing wired here degrades
597
+ every scenario to one retry-then-errored receipt rather than crashing the process."""
598
+
599
+
600
+ class NotWiredCallRunner:
601
+ async def run(
602
+ self,
603
+ scenario: Scenario,
604
+ runtime: EnvironmentRuntime,
605
+ *,
606
+ world: World | None = None,
607
+ ) -> CallOutcome:
608
+ del scenario, runtime, world
609
+ raise CallRunnerNotWired(
610
+ "no CallRunner wired -- the live voice-simulation call runner is a separate track"
611
+ )
612
+
613
+
614
+ _VOICE_CONNECTORS = {"livekit", "vapi", "retell"}
615
+
616
+
617
+ def _bundle_contract_value(bundle_dir: Path, key: str) -> str | None:
618
+ path = bundle_dir / "contract.json"
619
+ if not path.is_file():
620
+ return None
621
+ try:
622
+ body = json.loads(path.read_text(encoding="utf-8"))
623
+ except (OSError, ValueError):
624
+ return None
625
+ if not isinstance(body, dict):
626
+ return None
627
+ value = str(body.get(key) or "").strip().lower()
628
+ return value or None
629
+
630
+
631
+ def _bundle_contract_modality(bundle_dir: Path) -> str | None:
632
+ return _bundle_contract_value(bundle_dir, "modality")
633
+
634
+
635
+ def _default_build_call_runner(
636
+ adapter: "OutboundAdapter", context: CallRunnerContext
637
+ ) -> CallRunner:
638
+ """The real factory: `NotWiredCallRunner` stays exactly as documented for every connector
639
+ outside the LiveKit-dispatched voice path; a `"livekit"` job gets a real `CallRunnerImpl`,
640
+ whose OWN pre-dial validation (`call_runner._check_config`) is what surfaces an
641
+ incomplete-but-present config as a typed `call_failed`/infrastructure retry --
642
+ `capability_unavailable` stays unreachable from this seam (would require a scheduler edit;
643
+ the contract itself calls it "a follow-up, not shipped with this text")."""
644
+ connector = context.job.agent.connector.lower()
645
+ modality = _bundle_contract_modality(context.bundle_dir)
646
+ if connector == "retell_chat":
647
+ from .retell_chat_call_runner import RetellChatCallRunner
648
+
649
+ return RetellChatCallRunner(adapter, context)
650
+ if connector in _VOICE_CONNECTORS or (connector == "auto" and modality == "voice"):
651
+ # The understand stage read this off the agent's own instructions, so the contract is the
652
+ # only source. Carried through the process environment because `CallRunnerImpl` is handed a
653
+ # context and a scenario document, neither of which reaches the contract; this is an
654
+ # internal hop, not a knob, and nothing outside sets it.
655
+ declared = _bundle_contract_value(context.bundle_dir, "call_direction")
656
+ if declared:
657
+ os.environ["ALK_CALL_DIRECTION"] = declared
658
+ return CallRunnerImpl(adapter, context)
659
+ # Repository-hosted text targets advertise their concrete HTTP interface in the frozen
660
+ # contract adopted into Bundle V2. Connector-only Vapi/Retell remains on the existing
661
+ # NotWired path and is deliberately not inferred as repository chat.
662
+ if (context.bundle_dir / "contract.json").is_file():
663
+ from .chat_call_runner import HostedChatCallRunner
664
+
665
+ return HostedChatCallRunner(adapter, context)
666
+ return NotWiredCallRunner()
667
+
668
+
669
+ # =================================================================================================
670
+ # Scenario pre-allocation -- a thin client against endpoints.scenarios (outbound-channels.md v1.3
671
+ # Authentication: bearer + X-Harness-Fence, `{"result": {...}}` envelope, job-scoped idempotent).
672
+ # Previously unowned; owned by this module now.
673
+ # =================================================================================================
674
+
675
+
676
+ class ScenarioPreallocationError(RuntimeError):
677
+ def __init__(self, error: ob.ChannelError | None) -> None:
678
+ self.error = error
679
+ super().__init__(
680
+ "scenario pre-allocation failed" if error is None else error.message
681
+ )
682
+
683
+
684
+ class ScenariosClient:
685
+ """RESOLVED (p13-worker-r2, reports/p13-worker-r2.md CONTRACT NOTES): the Scenario
686
+ Generation Contract (PR #63) documented two paths (`run-tests/provision/` +
687
+ `run-tests/{id}/test-executions/`) and a position-ordered `scenario_ids` response, but the
688
+ platform's actual, live route (futureagi/simulate/views/hosted_harness.py:78-90,
689
+ urls.py:128-132) mints exactly ONE url per attempt -- a DRF detail `@action` with no
690
+ `url_path`, so the router only ever produces `.../scenarios/`, never a `provision/`/`begin/`
691
+ sub-resource. The real dispatch key is a body-level `operation: "provision"|"begin"` field
692
+ (serializers/hosted_harness.py:201-226's `HarnessScenarioOperationSerializer`). This class's
693
+ transport (`_post`) is unchanged -- `provision_path`/`begin_path` are the SAME
694
+ constructor-injectable placeholders as before, now correctly defaulted to an EMPTY suffix (the
695
+ real route needs none) rather than a guessed path segment; `register_with_platform`
696
+ (scenario_source.py) is what adds the `operation` field into each payload before calling
697
+ `.provision()`/`.begin()`, matching this class's existing "operation field in payload" seam
698
+ rather than requiring a change to either method's body. Shares `channel_state` with the other
699
+ three channels (a fence on any one must stop all of them, per outbound.py's own `ChannelState`
700
+ docstring)."""
701
+
702
+ def __init__(
703
+ self,
704
+ capabilities: ob.HostedCapabilities,
705
+ transport: ob.Transport | None = None,
706
+ *,
707
+ retry_policy: ob.RetryPolicy | None = None,
708
+ sleep: Callable[[float], None] = time.sleep,
709
+ rng: Callable[[], float] = random.random,
710
+ channel_state: ob.ChannelState | None = None,
711
+ provision_path: str = "",
712
+ begin_path: str = "",
713
+ ) -> None:
714
+ self._capabilities = capabilities
715
+ self._transport = transport or ob.RequestsTransport()
716
+ self._retry_policy = retry_policy or ob.RetryPolicy()
717
+ self._sleep = sleep
718
+ self._rng = rng
719
+ self._channel_state = channel_state or ob.ChannelState()
720
+ self._provision_path = provision_path
721
+ self._begin_path = begin_path
722
+
723
+ def provision(
724
+ self, payload: dict[str, Any], *, deadline: float | None = None
725
+ ) -> dict[str, Any]:
726
+ return self._post(self._provision_path, payload, deadline=deadline)
727
+
728
+ def begin(
729
+ self, payload: dict[str, Any], *, deadline: float | None = None
730
+ ) -> dict[str, Any]:
731
+ return self._post(self._begin_path, payload, deadline=deadline)
732
+
733
+ def _post(
734
+ self, path_suffix: str, payload: dict[str, Any], *, deadline: float | None
735
+ ) -> dict[str, Any]:
736
+ self._channel_state.check()
737
+ url = f"{self._capabilities.endpoints.scenarios}{path_suffix}"
738
+
739
+ def perform(_attempt: int) -> ob.TransportResponse:
740
+ return self._transport.request(
741
+ "POST",
742
+ url,
743
+ headers=self._capabilities.auth_headers(),
744
+ json_body=payload,
745
+ )
746
+
747
+ try:
748
+ response, error = ob._perform_with_retry(
749
+ perform,
750
+ retry_policy=self._retry_policy,
751
+ sleep=self._sleep,
752
+ rng=self._rng,
753
+ deadline=deadline,
754
+ )
755
+ except (ob.HostedFencedError, ob.HostedChannelFailedError) as exc:
756
+ self._channel_state.latch(exc)
757
+ raise
758
+ if error is not None or response is None:
759
+ raise ScenarioPreallocationError(error)
760
+ body = response.body if isinstance(response.body, dict) else {}
761
+ result = body.get("result")
762
+ if not isinstance(result, dict):
763
+ raise ScenarioPreallocationError(
764
+ ob.ChannelError(
765
+ ob.ChannelOutcome.PERMANENT_ITEM,
766
+ FailureDomain.PLATFORM_SYNC,
767
+ "scenarios_envelope_invalid",
768
+ "response body has no {'result': {...}} envelope",
769
+ )
770
+ )
771
+ return result
772
+
773
+
774
+ # =================================================================================================
775
+ # OutboundPort adapter -- the real emit pipeline: redact -> capabilities.event_builder() ->
776
+ # spool.append -> EventsClient.flush(). Also: baseline_frozen/parallelism_degraded from build.json,
777
+ # terminal events (exactly one, last), artifact-before-receipt ordering,
778
+ # and RunResult.aborted -> TerminalFailure(infrastructure, running, "world_pool_exhausted").
779
+ # =================================================================================================
780
+
781
+
782
+ _TERMINAL_FAILURE_MESSAGE_MAX_CHARS = 4096 # an unbounded `failure.message` can blow
783
+ # EVENT_PAYLOAD_MAX_BYTES and hard-reject the WHOLE terminal event; log is the only event type
784
+ # that self-truncates. 4KB is ample for a diagnostic message.
785
+
786
+
787
+ def _cap_failure_message(message: str) -> str:
788
+ if len(message) <= _TERMINAL_FAILURE_MESSAGE_MAX_CHARS:
789
+ return message
790
+ marker = "…[truncated]"
791
+ return message[: _TERMINAL_FAILURE_MESSAGE_MAX_CHARS - len(marker)] + marker
792
+
793
+
794
+ # guest-side mirror of outbound-channels.md's artifact level table (Channel 3) -- no module
795
+ # owns this table yet (the sealer's own version lives at `artifacts.py::seal_artifacts`, scoped to
796
+ # the local-SDK path); this hosted upload path needs its own "guest enforces it first" half.
797
+ _ARTIFACT_LEVEL_FORBIDDEN_KINDS: dict[ArtifactLevel, frozenset[ob.ArtifactKind]] = {
798
+ ArtifactLevel.METADATA_ONLY: frozenset(
799
+ {
800
+ ob.ArtifactKind.RECORDING_COMBINED,
801
+ ob.ArtifactKind.RECORDING_STEREO,
802
+ ob.ArtifactKind.RECORDING_CUSTOMER,
803
+ ob.ArtifactKind.RECORDING_ASSISTANT,
804
+ ob.ArtifactKind.TRACE,
805
+ ob.ArtifactKind.TOOL_TRACE,
806
+ ob.ArtifactKind.TRANSCRIPT,
807
+ ob.ArtifactKind.OTHER,
808
+ }
809
+ ),
810
+ ArtifactLevel.TRACES: frozenset(
811
+ {
812
+ ob.ArtifactKind.RECORDING_COMBINED,
813
+ ob.ArtifactKind.RECORDING_STEREO,
814
+ ob.ArtifactKind.RECORDING_CUSTOMER,
815
+ ob.ArtifactKind.RECORDING_ASSISTANT,
816
+ ob.ArtifactKind.OTHER,
817
+ }
818
+ ),
819
+ ArtifactLevel.TRACES_AND_RECORDINGS: frozenset({ob.ArtifactKind.OTHER}),
820
+ ArtifactLevel.FULL: frozenset(),
821
+ # `local-only` is rejected at hosted admission (`local_only_not_hosted`) per the contract --
822
+ # this adapter should never see it for a hosted job; forbid everything as a defensive default.
823
+ ArtifactLevel.LOCAL_ONLY: frozenset(ob.ArtifactKind),
824
+ }
825
+
826
+
827
+ class OutboundAdapter:
828
+ """Implements `hosted_scheduler.OutboundPort` plus the extra surface the entrypoint itself
829
+ needs (`stage_changed`, `baseline_frozen`, `parallelism_degraded`, `upload_artifact`,
830
+ `push_manifest`, `emit_terminal`) — hosted_scheduler.py only names the five methods scenario
831
+ code needs; everything else here is this module's own.
832
+
833
+ Fencing (HostedFencedError) is caught INTERNALLY by every method, never re-raised: letting it
834
+ escape into `HostedScheduler._emit()` (which catches bare `Exception` and tries to log through
835
+ the very port that just raised) would silently swallow the fence and let the scheduler keep
836
+ working an attempt that can no longer report anything. `is_fenced` is the flag the entrypoint's
837
+ orchestration (and `cancel_requested`) polls instead.
838
+ """
839
+
840
+ def __init__(
841
+ self,
842
+ capabilities: ob.HostedCapabilities,
843
+ *,
844
+ events_spool: ob.OutboundSpool,
845
+ events_client: ob.EventsClient,
846
+ results_client: ob.ResultsClient,
847
+ artifacts_client: ob.ArtifactsClient,
848
+ channel_state: ob.ChannelState,
849
+ extra_secret_values: tuple[str, ...] = (),
850
+ clock: Callable[[], datetime] = lambda: datetime.now(timezone.utc),
851
+ flush_window_seconds: float = ob.FLUSH_WINDOW_SECONDS,
852
+ ) -> None:
853
+ self._capabilities = capabilities
854
+ self._spool = events_spool
855
+ self._events = events_client
856
+ self._results = results_client
857
+ self._artifacts = artifacts_client
858
+ self._channel_state = channel_state
859
+ self._extra_secret_values = extra_secret_values
860
+ self._clock = clock
861
+ # `event_builder`'s own `extra_secret_values` binding is what lets
862
+ # `build_event_record` redact `log.message`/`world_unhealthy.cause`/
863
+ # `baseline_frozen.baseline_ref`/`terminal.failure.{code,message}` for every event this
864
+ # adapter emits -- binding it here, alongside identity, gives Channel 1 full redaction coverage.
865
+ self._event_builder = capabilities.event_builder(
866
+ extra_secret_values=extra_secret_values
867
+ )
868
+ self._stage_started = False
869
+ self._current_stage = HarnessStage.QUEUED
870
+ self._uploaded_digests: set[str] = set()
871
+ self._manifest_entries: list[dict[str, Any]] = []
872
+ self._terminal_emitted = False
873
+ # §0.6 v1.14 (exit code 4): the terminal record's own spool sequence, and whether the
874
+ # platform ever permanently rejected it by name -- `terminal_undelivered` (below) needs to
875
+ # tell "this specific record landed" apart from "some flush somewhere failed."
876
+ self._terminal_sequence: int | None = None
877
+ self._terminal_rejected = False
878
+ self._scenario_counts: dict[str, int] = {
879
+ "passed": 0,
880
+ "failed": 0,
881
+ "errored": 0,
882
+ "skipped": 0,
883
+ }
884
+ self._fenced_error: Exception | None = None
885
+ self._channel_failed_error: Exception | None = None
886
+ # the 120s flush window (§5.5) -- armed once, at whichever comes first: a cancel
887
+ # signal (`arm_flush_window` called explicitly by `run_job`'s `cancel_requested`) or the
888
+ # terminal event (`emit_terminal` below arms it itself, so no caller can forget).
889
+ self._flush_window_seconds = flush_window_seconds
890
+ self._flush_window_start: float | None = None
891
+ # job.artifacts is only known once job.json is parsed, which happens after this
892
+ # adapter is built (capabilities load, and the "no channel on a capabilities failure"
893
+ # contract, must come first) -- `configure_artifacts` below is called once it's available;
894
+ # this default is never actually exercised in practice, just a safe placeholder shape.
895
+ self._artifacts_policy = HarnessArtifactPolicy()
896
+ # `recording_headroom_bytes` stays 0 -- this adapter has no visibility into how many
897
+ # scenarios are still to run (or how large their recordings will be) at construction time,
898
+ # unlike the scheduler; sizing it here would be a guess dressed up as enforcement.
899
+ self._budget_tracker = ob.ArtifactBudgetTracker(
900
+ self._artifacts_policy.max_artifact_bytes
901
+ )
902
+ # `would_admit` (check) and `record` (reserve) must run as one atomic step -- two
903
+ # concurrent scenarios at W>1 racing the same remaining budget could otherwise both pass
904
+ # the check against a snapshot neither has updated yet.
905
+ self._artifact_budget_lock = asyncio.Lock()
906
+
907
+ @property
908
+ def is_fenced(self) -> bool:
909
+ return self._fenced_error is not None
910
+
911
+ @property
912
+ def terminal_undelivered(self) -> bool:
913
+ """The terminal was spooled (`emit_terminal` succeeded) but never confirmed delivered: the
914
+ platform permanently rejected the terminal item by name, or the spool's watermark never
915
+ reached the terminal's own sequence at all (channel exhaustion, a dead channel, or the
916
+ flush window running out before delivery). Exit 0 would claim a flush that provably never
917
+ happened. Fencing is checked by the caller first and always wins -- once fenced, whether
918
+ the terminal was ALSO undelivered is moot."""
919
+ if self._terminal_sequence is None or self.is_fenced:
920
+ return False
921
+ return (
922
+ self._terminal_rejected or self._spool.watermark() < self._terminal_sequence
923
+ )
924
+
925
+ @property
926
+ def scenario_counts(self) -> dict[str, int]:
927
+ return dict(self._scenario_counts)
928
+
929
+ def configure_artifacts(self, policy: HarnessArtifactPolicy) -> None:
930
+ self._artifacts_policy = policy
931
+ self._budget_tracker = ob.ArtifactBudgetTracker(policy.max_artifact_bytes)
932
+
933
+ def arm_flush_window(self) -> None:
934
+ if self._flush_window_start is None:
935
+ self._flush_window_start = time.monotonic()
936
+
937
+ def deadline(self) -> float | None:
938
+ if self._flush_window_start is None:
939
+ return None
940
+ return self._flush_window_start + self._flush_window_seconds
941
+
942
+ def _record_channel_error(self, exc: Exception) -> None:
943
+ if isinstance(exc, ob.HostedFencedError):
944
+ self._fenced_error = self._fenced_error or exc
945
+ else:
946
+ self._channel_failed_error = self._channel_failed_error or exc
947
+ logger.error("outbound channel latched: %s", exc)
948
+
949
+ def _guarded(self, fn: Callable[[], Any]) -> Any:
950
+ """Runs one outbound client call. Catches `HostedFencedError`/`HostedChannelFailedError`
951
+ so neither escapes as an ordinary exception (see class docstring)."""
952
+ try:
953
+ self._channel_state.check()
954
+ except (
955
+ ob.HostedFencedError,
956
+ ob.HostedChannelFailedError,
957
+ ob.HostedAttemptSupersededError,
958
+ ) as exc:
959
+ self._record_channel_error(exc)
960
+ return None
961
+ try:
962
+ return fn()
963
+ except (ob.HostedFencedError, ob.HostedChannelFailedError) as exc:
964
+ self._channel_state.latch(exc)
965
+ self._record_channel_error(exc)
966
+ return None
967
+
968
+ # -- events -------------------------------------------------------------------------------
969
+
970
+ def _emit_event(
971
+ self,
972
+ *,
973
+ stage: HarnessStage,
974
+ type_: ob.OutboundEventType,
975
+ payload: dict[str, Any],
976
+ ) -> None:
977
+ if self.is_fenced:
978
+ return # "stop emitting" -- no event of any type once fenced.
979
+ if self._terminal_emitted:
980
+ # `_bounded_close()` runs after the terminal is spooled -- an in-flight reconcile
981
+ # inside it can still call back into world_unhealthy/log. v1.3's "terminal ... exactly
982
+ # one, last emitted" is a hard invariant: anything after it is dropped locally, not
983
+ # spooled, rather than silently landing after the event the platform already finalized on.
984
+ # This also drops the rejected-event error log for anything the platform rejects on the
985
+ # SAME flush that carries the terminal -- diagnostic-quality only, since the rejected
986
+ # bytes are still recoverable as a `log`-kind artifact.
987
+ logger.warning(
988
+ "outbound event dropped after terminal: type=%s stage=%s",
989
+ type_.value,
990
+ stage.value,
991
+ )
992
+ return
993
+ event_id = f"event_{uuid.uuid4().hex}"
994
+ record = self._event_builder(
995
+ event_id=event_id,
996
+ emitted_at=self._clock(),
997
+ stage=stage,
998
+ type=type_,
999
+ payload=payload,
1000
+ )
1001
+ spooled = self._spool.append(record)
1002
+ if type_ is ob.OutboundEventType.TERMINAL:
1003
+ self._terminal_sequence = spooled.sequence
1004
+ self._current_stage = stage
1005
+
1006
+ async def _aemit_event(
1007
+ self,
1008
+ *,
1009
+ stage: HarnessStage,
1010
+ type_: ob.OutboundEventType,
1011
+ payload: dict[str, Any],
1012
+ ) -> None:
1013
+ # the spool append fsyncs the file AND its directory -- routed off the event loop so
1014
+ # it never stalls every other concurrently-running scenario at W>1.
1015
+ await asyncio.to_thread(
1016
+ self._emit_event, stage=stage, type_=type_, payload=payload
1017
+ )
1018
+
1019
+ def stage_changed(self, to: HarnessStage) -> None:
1020
+ frm = self._current_stage.value if self._stage_started else None
1021
+ self._stage_started = True
1022
+ observability.stage(to.value)
1023
+ self._emit_event(
1024
+ stage=to,
1025
+ type_=ob.OutboundEventType.STAGE_CHANGED,
1026
+ payload={"from": frm, "to": to.value},
1027
+ )
1028
+
1029
+ def baseline_frozen(self, *, inputs_digest: str, baseline_ref: str) -> None:
1030
+ self._emit_event(
1031
+ stage=HarnessStage.VALIDATING_ENVIRONMENT,
1032
+ type_=ob.OutboundEventType.BASELINE_FROZEN,
1033
+ payload={"inputs_digest": inputs_digest, "baseline_ref": baseline_ref},
1034
+ )
1035
+
1036
+ def parallelism_degraded(
1037
+ self, *, requested: int, effective: int, reason: str
1038
+ ) -> None:
1039
+ self._emit_event(
1040
+ stage=HarnessStage.VALIDATING_ENVIRONMENT,
1041
+ type_=ob.OutboundEventType.PARALLELISM_DEGRADED,
1042
+ payload={"requested": requested, "effective": effective, "reason": reason},
1043
+ )
1044
+
1045
+ def flush_events(
1046
+ self, *, deadline: float | None = None
1047
+ ) -> ob.EventsFlushResult | None:
1048
+ return self._guarded(lambda: self._events.flush(deadline=deadline))
1049
+
1050
+ async def aflush_events(self, *, deadline: float | None = None) -> None:
1051
+ result = await asyncio.to_thread(self.flush_events, deadline=deadline)
1052
+ # a rejected event's own payload never reaches the platform any other way -- surface
1053
+ # it via a `log` event (error) and keep the bytes recoverable as a `log`-kind artifact,
1054
+ # keyed off `dropped_records` (captured before the spool physically drops them).
1055
+ if result is None or not result.rejected:
1056
+ return
1057
+ if self._terminal_sequence is not None and any(
1058
+ entry.get("sequence") == self._terminal_sequence
1059
+ for entry in result.rejected
1060
+ ):
1061
+ # A permanent-item rejection is never retried -- the spool physically drops the record,
1062
+ # so no later flush can ever redeliver it.
1063
+ self._terminal_rejected = True
1064
+ dropped_by_sequence = {
1065
+ record.sequence: record for record in result.dropped_records
1066
+ }
1067
+ for entry in result.rejected:
1068
+ sequence = entry.get("sequence")
1069
+ await self.log(
1070
+ level="error",
1071
+ message=(
1072
+ f"event sequence={sequence} rejected by the platform: "
1073
+ f"{entry.get('code', 'unknown')}: {entry.get('message', '')}"
1074
+ ),
1075
+ )
1076
+ record = dropped_by_sequence.get(sequence)
1077
+ if record is not None:
1078
+ await self.upload_artifact(record.body, kind=ob.ArtifactKind.LOG)
1079
+
1080
+ # -- OutboundPort (hosted_scheduler.py) ----------------------------------------------------
1081
+
1082
+ async def scenario_started(
1083
+ self, *, scenario_key: str, world_index: int, scenario_attempt: int
1084
+ ) -> None:
1085
+ await self._aemit_event(
1086
+ stage=HarnessStage.RUNNING,
1087
+ type_=ob.OutboundEventType.SCENARIO_STARTED,
1088
+ payload={
1089
+ "scenario_key": scenario_key,
1090
+ "world_index": world_index,
1091
+ "scenario_attempt": scenario_attempt,
1092
+ },
1093
+ )
1094
+
1095
+ async def scenario_retried(
1096
+ self, *, scenario_key: str, from_world: int, to_world: int
1097
+ ) -> None:
1098
+ await self._aemit_event(
1099
+ stage=HarnessStage.RUNNING,
1100
+ type_=ob.OutboundEventType.SCENARIO_RETRIED,
1101
+ payload={
1102
+ "scenario_key": scenario_key,
1103
+ "from_world": from_world,
1104
+ "to_world": to_world,
1105
+ },
1106
+ )
1107
+
1108
+ async def world_unhealthy(self, *, world_index: int, cause: str) -> None:
1109
+ # world_unhealthy.cause <=200 (WorldUnhealthyPayload hard-rejects over that, so this must
1110
+ # truncate BEFORE `_aemit_event`, not rely on the builder's own redaction, which runs
1111
+ # after this call and could not shrink an already-too-long string back into budget).
1112
+ redacted = ob.redact_outbound_text(cause, self._extra_secret_values)
1113
+ if len(redacted) > 200:
1114
+ redacted = redacted[:200]
1115
+ await self._aemit_event(
1116
+ stage=HarnessStage.RUNNING,
1117
+ type_=ob.OutboundEventType.WORLD_UNHEALTHY,
1118
+ payload={"world_index": world_index, "cause": redacted},
1119
+ )
1120
+
1121
+ async def log(self, *, level: str, message: str) -> None:
1122
+ await self._aemit_event(
1123
+ stage=self._current_stage,
1124
+ type_=ob.OutboundEventType.LOG,
1125
+ payload={"level": level, "message": message},
1126
+ )
1127
+
1128
+ async def receipt(self, receipt: ResultReceipt) -> None:
1129
+ if self.is_fenced:
1130
+ return
1131
+ # counted only once a push is actually attempted -- counting before this point would
1132
+ # include receipts that were never pushed (and the counts feed the terminal payload).
1133
+ self._scenario_counts[receipt.status] = (
1134
+ self._scenario_counts.get(receipt.status, 0) + 1
1135
+ )
1136
+ call: dict[str, Any] | None = None
1137
+ if receipt.call is not None and receipt.call.started_at is not None:
1138
+ transcript_artifact = receipt.call.transcript_artifact
1139
+ if transcript_artifact is not None:
1140
+ bare = transcript_artifact.split(":", 1)[-1]
1141
+ if bare not in self._uploaded_digests:
1142
+ # null it rather than shipping a receipt the platform will 422
1143
+ # (`artifact_unknown`) wholesale -- the contract explicitly blesses a null
1144
+ # `transcript_artifact` "named in a log event."
1145
+ await self.log(
1146
+ level="error",
1147
+ message=(
1148
+ f"receipt for {receipt.scenario_key} references un-acked transcript "
1149
+ f"artifact {transcript_artifact}; nulling it"
1150
+ ),
1151
+ )
1152
+ transcript_artifact = None
1153
+ recording_artifacts: list[str] = []
1154
+ for artifact_id in receipt.call.recording_artifacts:
1155
+ bare = artifact_id.split(":", 1)[-1] if artifact_id else None
1156
+ if artifact_id and bare not in self._uploaded_digests:
1157
+ await self.log(
1158
+ level="error",
1159
+ message=(
1160
+ f"receipt for {receipt.scenario_key} references un-acked recording "
1161
+ f"artifact {artifact_id}; dropping it"
1162
+ ),
1163
+ )
1164
+ continue
1165
+ recording_artifacts.append(artifact_id)
1166
+ ended_at = receipt.call.ended_at
1167
+ if ended_at is None:
1168
+ # outbound.CallSummary.ended_at is a required str -- a call that started but
1169
+ # never finished (CallAborted's partial) would otherwise fail build_result_receipt's
1170
+ # validation and silently drop the whole receipt (HostedScheduler._emit's blanket
1171
+ # except swallows it).
1172
+ ended_at = receipt.call.started_at
1173
+ await self.log(
1174
+ level="warning",
1175
+ message=(
1176
+ f"receipt for {receipt.scenario_key} has no call.ended_at; substituting "
1177
+ "started_at"
1178
+ ),
1179
+ )
1180
+ call = {
1181
+ "started_at": receipt.call.started_at,
1182
+ "ended_at": ended_at,
1183
+ "duration_ms": receipt.call.duration_ms,
1184
+ "turns": receipt.call.turns,
1185
+ "transcript_artifact": transcript_artifact,
1186
+ "recording_artifacts": recording_artifacts,
1187
+ }
1188
+ stop_reason = getattr(receipt.call, "stop_reason", None)
1189
+ if stop_reason:
1190
+ call["stop_reason"] = stop_reason
1191
+ elif receipt.call is not None:
1192
+ # `hosted_scheduler.CallSummary.started_at` is `str | None`, but
1193
+ # `outbound.CallSummary.started_at` requires a real timestamp -- per the contract a
1194
+ # call summary is only present once the call has genuinely started, so a call that
1195
+ # never started is omitted here rather than shipped with a value that would fail
1196
+ # `build_result_receipt`'s own validation.
1197
+ await self.log(
1198
+ level="warning",
1199
+ message=f"receipt for {receipt.scenario_key} has no call.started_at; omitting call",
1200
+ )
1201
+ failure: dict[str, Any] | None = None
1202
+ if receipt.failure is not None:
1203
+ # Redact before capping -- truncating first can cut a secret in half at
1204
+ # the boundary and leave exact-substring redaction unable to find the surviving piece.
1205
+ redacted_failure_message = ob.redact_outbound_text(
1206
+ receipt.failure.message, self._extra_secret_values
1207
+ )
1208
+ failure = {
1209
+ "domain": receipt.failure.domain,
1210
+ "stage": receipt.failure.stage,
1211
+ "code": receipt.failure.code,
1212
+ "message": _cap_failure_message(redacted_failure_message),
1213
+ }
1214
+ wire = ob.build_result_receipt(
1215
+ job_id=self._capabilities.job_id,
1216
+ attempt_id=self._capabilities.attempt_id,
1217
+ attempt_number=self._capabilities.attempt_number,
1218
+ scenario_key=receipt.scenario_key,
1219
+ scenario_id=receipt.scenario_id,
1220
+ scenario_attempt=receipt.scenario_attempt,
1221
+ world_index=receipt.world_index,
1222
+ status=receipt.status,
1223
+ sub_goals=[
1224
+ {"name": g.name, "held": g.held, "reason": g.reason, "judged": g.judged}
1225
+ for g in receipt.sub_goals
1226
+ ],
1227
+ evaluations=[_evaluation_wire(e) for e in receipt.evaluations],
1228
+ call=call,
1229
+ failure=failure,
1230
+ extra_secret_values=self._extra_secret_values,
1231
+ )
1232
+ push_result = await asyncio.to_thread(
1233
+ self._guarded, lambda: self._results.push(wire)
1234
+ )
1235
+ if push_result is not None and push_result.error is not None:
1236
+ # The contract's own obligation for a permanent rejection (e.g. 409 receipt_conflict,
1237
+ # 422 artifact_unknown): "the platform keeps the first; guest logs, no retry." `push()`
1238
+ # returns this rather than raising, so nothing inspected it before now.
1239
+ await self.log(
1240
+ level="error",
1241
+ message=(
1242
+ f"receipt for {receipt.scenario_key} rejected by the platform: "
1243
+ f"{push_result.error.code}: {push_result.error.message}"
1244
+ ),
1245
+ )
1246
+ await self.aflush_events()
1247
+
1248
+ # -- artifacts (uploaded+acked BEFORE the referencing receipt) --------------------------
1249
+
1250
+ async def upload_artifact(
1251
+ self,
1252
+ data: bytes,
1253
+ *,
1254
+ kind: ob.ArtifactKind,
1255
+ scenario_key: str | None = None,
1256
+ deadline: float | None = None,
1257
+ ) -> str | None:
1258
+ """Returns the `sha256:<64-hex>` id form the wire uses (`CallSummary.transcript_artifact`,
1259
+ `ArtifactManifestEntry.artifact_id`) — never the bare hex `ArtifactsClient.upload` itself
1260
+ takes, which is a different, easy-to-mix-up shape (this module's own report notes it)."""
1261
+ digest = hashlib.sha256(data).hexdigest()
1262
+ if digest in self._uploaded_digests:
1263
+ return f"sha256:{digest}"
1264
+ if self.is_fenced:
1265
+ return None
1266
+ # guest-side level admission + budget, both BEFORE the transport is ever touched
1267
+ # ("the guest enforces it first").
1268
+ forbidden = _ARTIFACT_LEVEL_FORBIDDEN_KINDS.get(
1269
+ self._artifacts_policy.level, frozenset()
1270
+ )
1271
+ if kind in forbidden:
1272
+ await self.log(
1273
+ level="error",
1274
+ message=(
1275
+ f"artifact upload refused: kind={kind.value} forbidden at "
1276
+ f"level={self._artifacts_policy.level.value}"
1277
+ ),
1278
+ )
1279
+ return None
1280
+ # check-and-reserve atomically, before the actual (slow, concurrency-safe) upload --
1281
+ # see the lock's own comment in __init__. A failed upload below leaves the reservation in
1282
+ # place rather than releasing it (`ArtifactBudgetTracker` has no release primitive): a
1283
+ # stuck-conservative budget is safe, an under-counted one that lets two racing uploads both
1284
+ # pass admission is not.
1285
+ async with self._artifact_budget_lock:
1286
+ if not self._budget_tracker.would_admit(kind, len(data), digest=digest):
1287
+ await self.log(
1288
+ level="error",
1289
+ message=(
1290
+ f"artifact upload refused: budget exhausted (kind={kind.value}, "
1291
+ f"size={len(data)})"
1292
+ ),
1293
+ )
1294
+ return None
1295
+ self._budget_tracker.record(kind, len(data), digest=digest)
1296
+ result = await asyncio.to_thread(
1297
+ self._guarded,
1298
+ lambda: self._artifacts.upload(
1299
+ digest, data, kind=kind, scenario_key=scenario_key, deadline=deadline
1300
+ ),
1301
+ )
1302
+ if result is None or result.error is not None:
1303
+ code = (
1304
+ result.error.code
1305
+ if result is not None and result.error is not None
1306
+ else "fenced"
1307
+ )
1308
+ await self.log(
1309
+ level="error", message=f"artifact upload failed ({kind.value}): {code}"
1310
+ )
1311
+ return None
1312
+ self._uploaded_digests.add(digest)
1313
+ self._manifest_entries.append(
1314
+ {
1315
+ "artifact_id": f"sha256:{digest}",
1316
+ "kind": kind.value,
1317
+ "size": len(data),
1318
+ "scenario_key": scenario_key,
1319
+ }
1320
+ )
1321
+ return f"sha256:{digest}"
1322
+
1323
+ async def push_manifest(
1324
+ self, *, complete: bool, deadline: float | None = None
1325
+ ) -> bool:
1326
+ if self.is_fenced:
1327
+ return False
1328
+ wire = ob.build_artifact_manifest(
1329
+ job_id=self._capabilities.job_id,
1330
+ attempt_id=self._capabilities.attempt_id,
1331
+ attempt_number=self._capabilities.attempt_number,
1332
+ entries=list(self._manifest_entries),
1333
+ complete=complete,
1334
+ )
1335
+ result = await asyncio.to_thread(
1336
+ self._guarded,
1337
+ lambda: self._artifacts.push_manifest(wire, deadline=deadline),
1338
+ )
1339
+ if result is None:
1340
+ return False
1341
+ if not result.delivered:
1342
+ error = result.error
1343
+ logger.error(
1344
+ "artifact manifest delivery failed: code=%s message=%s",
1345
+ error.code if error is not None else "unknown",
1346
+ error.message if error is not None else "no response",
1347
+ )
1348
+ return False
1349
+ return True
1350
+
1351
+ async def ensure_terminal_artifacts(
1352
+ self,
1353
+ *,
1354
+ work_directory: Path,
1355
+ stage: HarnessStage,
1356
+ failure: dict[str, Any] | None,
1357
+ ) -> None:
1358
+ """Upload the three artifacts a complete platform manifest requires.
1359
+
1360
+ The platform contract requires ``build``, ``result`` and ``log`` even when a run has no
1361
+ transcript/recording. Previously the adapter only uploaded a log when an event was
1362
+ rejected, so every otherwise-successful complete manifest was deterministically rejected.
1363
+ """
1364
+ build_path = work_directory / "build.json"
1365
+ build = (
1366
+ build_path.read_bytes()
1367
+ if build_path.is_file()
1368
+ else b'{"status":"build metadata unavailable"}\n'
1369
+ )
1370
+ result = (
1371
+ json.dumps(
1372
+ {
1373
+ "stage": stage.value,
1374
+ "failure": failure,
1375
+ "scenario_counts": self.scenario_counts,
1376
+ },
1377
+ sort_keys=True,
1378
+ separators=(",", ":"),
1379
+ ).encode()
1380
+ + b"\n"
1381
+ )
1382
+ log = (
1383
+ f"hosted harness terminal stage={stage.value}; "
1384
+ f"scenario_counts={json.dumps(self.scenario_counts, sort_keys=True)}\n"
1385
+ ).encode()
1386
+ for body, kind in (
1387
+ (build, ob.ArtifactKind.BUILD),
1388
+ (result, ob.ArtifactKind.RESULT),
1389
+ (log, ob.ArtifactKind.LOG),
1390
+ ):
1391
+ await self.upload_artifact(body, kind=kind, deadline=self.deadline())
1392
+
1393
+ # -- terminal (exactly one terminal event, last emitted) --------------------------------
1394
+
1395
+ async def emit_terminal(
1396
+ self,
1397
+ *,
1398
+ stage: HarnessStage,
1399
+ reason: ob.TerminalReason | None = None,
1400
+ failure: dict[str, Any] | None = None,
1401
+ ) -> bool:
1402
+ """Returns whether a terminal event was actually emitted (False when already emitted, or
1403
+ fenced). No caller reads this return value any more -- `drain()`'s own manifest push is
1404
+ gated on `is_fenced` instead; kept `bool` since a future caller may still want it."""
1405
+ if self._terminal_emitted or self.is_fenced:
1406
+ return False
1407
+ if failure is not None and isinstance(failure.get("message"), str):
1408
+ # redact BEFORE truncating -- the inverse order can cut a secret in half at the 4KB
1409
+ # boundary, and exact-substring redaction can no longer find the surviving fragment.
1410
+ redacted = ob.redact_outbound_text(
1411
+ failure["message"], self._extra_secret_values
1412
+ )
1413
+ failure = {**failure, "message": _cap_failure_message(redacted)}
1414
+ # the latch is set AFTER a successful append (below), not before -- a raise inside
1415
+ # `_emit_event` (an oversized payload, an invalid `failure.domain`) must not permanently
1416
+ # disable the terminal event.
1417
+ self._emit_event(
1418
+ stage=stage,
1419
+ type_=ob.OutboundEventType.TERMINAL,
1420
+ payload={
1421
+ "stage": stage.value,
1422
+ "reason": reason.value if reason is not None else None,
1423
+ "failure": failure,
1424
+ "scenario_counts": dict(self._scenario_counts),
1425
+ },
1426
+ )
1427
+ self._terminal_emitted = True
1428
+ # arm the flush window HERE, unconditionally -- "120s from the cancel signal / TTL /
1429
+ # terminal event," not only when a cancel was separately observed.
1430
+ self.arm_flush_window()
1431
+ return True
1432
+
1433
+ async def flush_terminal(self, *, deadline: float | None = None) -> bool:
1434
+ """`emit_terminal` only appends the terminal record to the LOCAL spool -- a
1435
+ caller that pushes something else on the wire right after (skipped receipts) would
1436
+ otherwise risk `receipt()`'s own trailing `aflush_events()` delivering the terminal as a
1437
+ side effect of pushing THAT receipt, landing the terminal after it on the wire. Same
1438
+ bounded loop as `drain()` (a backlog bigger than one `EVENTS_MAX_BATCH` must not
1439
+ strand the terminal), stopping short of the manifest push -- that still belongs after
1440
+ skipped receipts, not here. Returns whether a fence was observed."""
1441
+ while True:
1442
+ before = self._spool.watermark()
1443
+ await self.aflush_events(deadline=deadline)
1444
+ if self.is_fenced or not self._spool.pending_since_watermark():
1445
+ break
1446
+ if self._spool.watermark() == before or (
1447
+ deadline is not None and time.monotonic() >= deadline
1448
+ ):
1449
+ break
1450
+ return self.is_fenced
1451
+
1452
+ async def drain(self, *, complete: bool, deadline: float | None = None) -> bool:
1453
+ """Best-effort final delivery: events, then the artifact manifest (Sequencing: "terminal
1454
+ event -> receipts (incl. synthesized skipped) -> manifest" -- receipts are already pushed
1455
+ individually by `receipt()` as each scenario finishes). Returns whether a fence was
1456
+ observed during (or before) this call -- the fence most often lands on the very flush
1457
+ that carries the terminal event, so the caller's exit code must come from THIS return
1458
+ value, never a fence check taken before drain() ran.
1459
+
1460
+ `aflush_events` delivers at most ONE `EVENTS_MAX_BATCH`-sized batch per call -- a
1461
+ backlog bigger than that (the rejected-event logging in `aflush_events` can grow one) would otherwise
1462
+ strand the terminal event, the highest sequence, undelivered while still exiting 0. Loops
1463
+ until the spool is actually empty, a fence is observed, a flush makes no further progress,
1464
+ or the deadline is gone -- whichever comes first.
1465
+ """
1466
+ while True:
1467
+ before = self._spool.watermark()
1468
+ await self.aflush_events(deadline=deadline)
1469
+ if self.is_fenced or not self._spool.pending_since_watermark():
1470
+ break
1471
+ if self._spool.watermark() == before or (
1472
+ deadline is not None and time.monotonic() >= deadline
1473
+ ):
1474
+ break
1475
+ await self.push_manifest(complete=complete, deadline=deadline)
1476
+ return self.is_fenced
1477
+
1478
+
1479
+ def _evaluation_wire(evaluation: Any) -> dict[str, Any]:
1480
+ if evaluation.kind == "metric":
1481
+ return {
1482
+ # `MetricEvaluation.score: float` coerces `1` -> `1.0`; `build_result_receipt`
1483
+ # digests the RAW dict before that coercion, so an int here would digest-mismatch
1484
+ # against the model's own re-derivation and silently drop the receipt.
1485
+ "name": evaluation.name,
1486
+ "kind": "metric",
1487
+ "score": float(evaluation.score) if evaluation.score is not None else None,
1488
+ "reason": evaluation.reason,
1489
+ }
1490
+ return {
1491
+ "name": evaluation.name,
1492
+ "kind": "checkpoint",
1493
+ "passed": evaluation.passed,
1494
+ "reason": evaluation.reason,
1495
+ }
1496
+
1497
+
1498
+ # =================================================================================================
1499
+ # Cancellation -- spine §0 step 7 / outbound-channels.md "Cancellation signal": the gateway writes
1500
+ # `cancel_path` then sends SIGTERM; the guest stops LAUNCHING new scenarios (not killing what's
1501
+ # already running) and starts the 120s flush window.
1502
+ # =================================================================================================
1503
+
1504
+
1505
+ class CancelState:
1506
+ def __init__(self, path: Path) -> None:
1507
+ self._path = path
1508
+ self._sigterm_seen = False
1509
+
1510
+ def note_sigterm(self) -> None:
1511
+ self._sigterm_seen = True
1512
+
1513
+ def requested(self) -> bool:
1514
+ return self._sigterm_seen or self._path.exists()
1515
+
1516
+ def reason(self) -> ob.TerminalReason | None:
1517
+ try:
1518
+ raw = json.loads(self._path.read_text(encoding="utf-8"))
1519
+ except (OSError, ValueError):
1520
+ return None
1521
+ value = raw.get("reason") if isinstance(raw, dict) else None
1522
+ try:
1523
+ return ob.TerminalReason(value)
1524
+ except ValueError:
1525
+ return None
1526
+
1527
+
1528
+ def install_sigterm_handler(cancel_state: CancelState) -> Callable[[], None]:
1529
+ """Best-effort: `signal.signal` only works on the process's main thread and raises
1530
+ `ValueError` anywhere else (e.g. inside a test running on a worker thread) — caught and
1531
+ turned into a no-op restore, since `cancel_requested` still works off the file alone."""
1532
+
1533
+ def _handler(signum: int, frame: Any) -> None:
1534
+ del signum, frame
1535
+ cancel_state.note_sigterm()
1536
+
1537
+ try:
1538
+ previous = signal.signal(signal.SIGTERM, _handler)
1539
+ except (ValueError, OSError):
1540
+ return lambda: None
1541
+
1542
+ def _restore() -> None:
1543
+ try:
1544
+ signal.signal(signal.SIGTERM, previous)
1545
+ except (ValueError, OSError):
1546
+ pass
1547
+
1548
+ return _restore
1549
+
1550
+
1551
+ def default_install_sigterm_handler(cancel_state: CancelState) -> Callable[[], None]:
1552
+ return install_sigterm_handler(cancel_state)
1553
+
1554
+
1555
+ def _resolve_hosted_public_url(
1556
+ capabilities: ob.HostedCapabilities,
1557
+ transport: ob.Transport,
1558
+ *,
1559
+ port: int,
1560
+ expires_in_seconds: int,
1561
+ ) -> str:
1562
+ endpoint = capabilities.endpoints.ingress
1563
+ if not endpoint:
1564
+ raise ProcessRuntimeError(
1565
+ "provider_lifecycle",
1566
+ "spawn_failed",
1567
+ "the platform did not grant an ingress capability for provider webhooks",
1568
+ domain=FailureDomain.INFRASTRUCTURE,
1569
+ )
1570
+ try:
1571
+ response = transport.request(
1572
+ "POST",
1573
+ endpoint,
1574
+ headers=capabilities.auth_headers(),
1575
+ json_body={
1576
+ "port": port,
1577
+ "expires_in_seconds": expires_in_seconds,
1578
+ },
1579
+ timeout=30.0,
1580
+ )
1581
+ except ob.TransportError as exc:
1582
+ raise ProcessRuntimeError(
1583
+ "provider_lifecycle",
1584
+ "spawn_failed",
1585
+ f"platform ingress request failed: {exc}",
1586
+ domain=FailureDomain.INFRASTRUCTURE,
1587
+ ) from exc
1588
+ body = response.body or {}
1589
+ url = body.get("url")
1590
+ if (
1591
+ response.status_code != 200
1592
+ or not isinstance(url, str)
1593
+ or not url.startswith("https://")
1594
+ ):
1595
+ raise ProcessRuntimeError(
1596
+ "provider_lifecycle",
1597
+ "spawn_failed",
1598
+ f"platform ingress request was rejected with HTTP {response.status_code}",
1599
+ domain=FailureDomain.INFRASTRUCTURE,
1600
+ )
1601
+ return url
1602
+
1603
+
1604
+ # =================================================================================================
1605
+ # Dependency injection -- every seam a test needs to replace with a fake, gathered in one place so
1606
+ # `run_job` itself stays pure orchestration.
1607
+ # =================================================================================================
1608
+
1609
+
1610
+ @dataclass
1611
+ class HostedEntrypointDeps:
1612
+ load_capabilities: Callable[[], ob.HostedCapabilities] = field(
1613
+ default=lambda: ob.load_capabilities()
1614
+ )
1615
+ bundle_source: BundleSource = field(default_factory=DefaultBundleSource)
1616
+ scenario_source: ScenarioSource = field(default_factory=NotWiredScenarioSource)
1617
+ build_transport: Callable[[], ob.Transport] = field(
1618
+ default=lambda: ob.RequestsTransport()
1619
+ )
1620
+ # Daytona forces the sandbox to a fixed non-root user (svc-control) and ignores os_user
1621
+ # overrides, so the guest cannot setuid/chown to the bundle's svc-agent/svc-tools/svc-data
1622
+ # users -- every process runs uniformly as svc-control. The bundle may still DECLARE those
1623
+ # users (the model validates them); they are simply not enforced at runtime here.
1624
+ build_provider: Callable[
1625
+ [ob.HostedCapabilities, ob.Transport], WorldProvisioner
1626
+ ] = field(
1627
+ default=lambda capabilities, transport: ProcessRuntimeProvider(
1628
+ user_resolver=lambda _name: None,
1629
+ require_declared_user=False,
1630
+ public_url_resolver=lambda port, ttl: _resolve_hosted_public_url(
1631
+ capabilities, transport, port=port, expires_in_seconds=ttl
1632
+ ),
1633
+ provider_attempt_id=capabilities.attempt_id,
1634
+ provider_expires_at=capabilities.expires_at,
1635
+ )
1636
+ )
1637
+ # The real call runner needs `OutboundAdapter.upload_artifact` to satisfy the invariant that
1638
+ # referenced artifacts are uploaded+acked BEFORE the receipt that names them -- the adapter is
1639
+ # threaded in once `run_job` has built it, rather than the CallRunner reaching for a global.
1640
+ # `CallRunnerContext` carries everything else `CallRunnerImpl` needs (job, bundle_dir,
1641
+ # evidence_seam, the target_provider secret map, attempt_number) that the bare
1642
+ # `CallRunner.run(scenario, runtime)` protocol has no room for.
1643
+ build_call_runner: Callable[["OutboundAdapter", CallRunnerContext], CallRunner] = (
1644
+ field(
1645
+ default=lambda adapter, context: _default_build_call_runner(
1646
+ adapter, context
1647
+ )
1648
+ )
1649
+ )
1650
+ build_world_factory: Callable[[Path], WorldFactory] = field(
1651
+ default=ProcessWorldFactory
1652
+ )
1653
+ retry_policy: Callable[[], ob.RetryPolicy] = field(default=lambda: ob.RetryPolicy())
1654
+ clock: Callable[[], datetime] = field(default=lambda: datetime.now(timezone.utc))
1655
+ cancel_path: Path = field(default_factory=lambda: Path(CANCEL_SIGNAL_PATH))
1656
+ secrets_path: Path = field(default_factory=lambda: SECRETS_PATH)
1657
+ simulator_secrets_path: Path = field(default_factory=lambda: SIMULATOR_SECRETS_PATH)
1658
+ flush_window_seconds: float = ob.FLUSH_WINDOW_SECONDS
1659
+ install_sigterm_handler: Callable[[CancelState], Callable[[], None]] = field(
1660
+ default=default_install_sigterm_handler
1661
+ )
1662
+ events_spool_dir_name: str = EVENTS_SPOOL_DIR_NAME
1663
+ scenarios_client_kwargs: dict[str, Any] = field(default_factory=dict)
1664
+
1665
+ def build_events_spool(self, work_directory: Path) -> ob.OutboundSpool:
1666
+ return ob.OutboundSpool(
1667
+ work_directory / self.events_spool_dir_name, "events", sequenced=True
1668
+ )
1669
+
1670
+ def build_scenarios_client(
1671
+ self,
1672
+ capabilities: ob.HostedCapabilities,
1673
+ transport: ob.Transport,
1674
+ channel_state: ob.ChannelState,
1675
+ ) -> ScenariosClient:
1676
+ return ScenariosClient(
1677
+ capabilities,
1678
+ transport,
1679
+ channel_state=channel_state,
1680
+ **self.scenarios_client_kwargs,
1681
+ )
1682
+
1683
+ def peek_secret_values(self) -> tuple[str, ...]:
1684
+ return peek_secret_values(self.secrets_path)
1685
+
1686
+ def peek_target_provider_secret_values(
1687
+ self, secret_purposes: dict[str, str]
1688
+ ) -> dict[str, str]:
1689
+ return peek_target_provider_secret_values(self.secrets_path, secret_purposes)
1690
+
1691
+ def peek_simulator_provider_secret_values(
1692
+ self, secret_purposes: dict[str, str]
1693
+ ) -> dict[str, str]:
1694
+ return peek_simulator_provider_secret_values(self.secrets_path, secret_purposes)
1695
+
1696
+ def load_simulator_secret_values(self) -> dict[str, str]:
1697
+ return load_simulator_secret_values(self.simulator_secrets_path)
1698
+
1699
+
1700
+ # =================================================================================================
1701
+ # Scenario-entry validation at fetch (defense against karthik-integration-changes.md K1): the
1702
+ # Scenario Generation Contract's own model may not carry `scenario_key` (or may hand back some
1703
+ # other malformed shape) by the time `scenario_source.build()` returns it here, and
1704
+ # `hosted_scheduler.py` reads `scenario.scenario_key`/`.sub_goals`/`.setup`/`.ready` at its own
1705
+ # call sites with plain attribute access -- an attribute a pydantic/dataclass model never defined
1706
+ # raises AttributeError, not a typed failure, deep inside the scheduler with no terminal event and
1707
+ # a nonzero exit that reads as an infrastructure crash. Checked here with `getattr` (never direct
1708
+ # attribute access) so a malformed entry is caught at the seam, before the scheduler ever touches
1709
+ # it -- one bad entry fails the whole job as a typed FAILED terminal instead of crashing the guest.
1710
+ # =================================================================================================
1711
+
1712
+ # No closed-vocabulary code names this defect specifically (the §2e/§2f tables are bundle/process
1713
+ # concerns, not scenario-content ones) -- `scenario_preallocation_failed` is this module's own
1714
+ # existing code for "the scenario set is not viable for this attempt," already scoped to stage
1715
+ # `validating_scenarios`, and is reused here rather than inventing a new one. Domain `environment`
1716
+ # (not `platform_sync`, its other use here): a malformed entry is a deterministic generation-stage
1717
+ # content defect, not a transport failure, and fails identically on retry.
1718
+ _SCENARIO_ENTRY_INVALID_CODE = "scenario_preallocation_failed"
1719
+
1720
+
1721
+ def _validate_scenario_entry(entry: Any, *, index: int) -> str | None:
1722
+ """Returns a human-readable defect description, or `None` if `entry` looks usable by
1723
+ `hosted_scheduler.py`'s `Scenario` Protocol. Every check is a `getattr` with a default, never
1724
+ a direct attribute/index access -- the whole point is to survive a shape that lacks a field
1725
+ entirely, not just one that carries a wrong value.
1726
+ """
1727
+ scenario_key = getattr(entry, "scenario_key", None)
1728
+ if not isinstance(scenario_key, str) or not scenario_key:
1729
+ return f"scenario[{index}] has no non-empty scenario_key"
1730
+ label = f"scenario[{index}] ({scenario_key!r})"
1731
+ if not isinstance(getattr(entry, "scenario_id", None), str):
1732
+ return f"{label} has no scenario_id"
1733
+ if not callable(getattr(entry, "setup", None)):
1734
+ return f"{label} has no callable setup()"
1735
+ if not callable(getattr(entry, "ready", None)):
1736
+ return f"{label} has no callable ready()"
1737
+ sub_goals = getattr(entry, "sub_goals", None)
1738
+ if not isinstance(sub_goals, Sequence) or isinstance(sub_goals, (str, bytes)):
1739
+ return f"{label} has no sub_goals sequence"
1740
+ for goal_index, goal in enumerate(sub_goals):
1741
+ goal_name = getattr(goal, "name", None)
1742
+ if not isinstance(goal_name, str) or not goal_name:
1743
+ return f"{label} sub_goal[{goal_index}] has no non-empty name"
1744
+ if not callable(getattr(goal, "check", None)):
1745
+ return f"{label} sub_goal[{goal_index}] ({goal_name!r}) has no callable check()"
1746
+ return None
1747
+
1748
+
1749
+ def _validate_scenarios(scenarios: Sequence[Any]) -> str | None:
1750
+ for index, entry in enumerate(scenarios):
1751
+ defect = _validate_scenario_entry(entry, index=index)
1752
+ if defect is not None:
1753
+ return defect
1754
+ return None
1755
+
1756
+
1757
+ # =================================================================================================
1758
+ # Orchestration -- steps 1-8, in order.
1759
+ # =================================================================================================
1760
+
1761
+
1762
+ async def run_job(
1763
+ job_path: Path,
1764
+ source: Path,
1765
+ output: Path,
1766
+ *,
1767
+ deps: HostedEntrypointDeps | None = None,
1768
+ ) -> int:
1769
+ """The guest's whole `main()` body. Returns the process exit code (§0.6) — `main()` below is
1770
+ the only caller that turns this into `SystemExit`, so tests can call this directly and assert
1771
+ on the return value."""
1772
+ deps = deps or HostedEntrypointDeps()
1773
+ work_directory = output.parent
1774
+
1775
+ # This control-process-only channel is loaded before any Stage or CallRunner is constructed.
1776
+ # It is separate from secrets.json so platform simulator credentials never acquire the
1777
+ # ``target_provider`` purpose and therefore can never enter an agent process.
1778
+ simulator_secret_values = deps.load_simulator_secret_values()
1779
+ os.environ.update(simulator_secret_values)
1780
+
1781
+ # 1. Boot -- capabilities. CapabilitiesError -> exit non-zero-and-NOT-3, no event (v1.3 table):
1782
+ # there is no channel yet to report a terminal event through.
1783
+ try:
1784
+ capabilities = deps.load_capabilities()
1785
+ except ob.CapabilitiesError as exc:
1786
+ logger.error("capabilities load failed: %s: %s", exc.code, exc.message)
1787
+ return EXIT_BOOT_FAILURE
1788
+
1789
+ # Every line from here on is attributable. Done as early as the id is known, which is
1790
+ # immediately after capabilities load.
1791
+ configure_runner_logging(getattr(capabilities, "job_id", None))
1792
+
1793
+ channel_state = ob.ChannelState()
1794
+ transport = deps.build_transport()
1795
+ retry_policy = deps.retry_policy()
1796
+ events_spool = deps.build_events_spool(work_directory)
1797
+ events_client = ob.EventsClient(
1798
+ capabilities,
1799
+ events_spool,
1800
+ transport,
1801
+ retry_policy=retry_policy,
1802
+ channel_state=channel_state,
1803
+ )
1804
+ results_client = ob.ResultsClient(
1805
+ capabilities, transport, retry_policy=retry_policy, channel_state=channel_state
1806
+ )
1807
+ artifacts_client = ob.ArtifactsClient(
1808
+ capabilities, transport, retry_policy=retry_policy, channel_state=channel_state
1809
+ )
1810
+ scenarios_client = deps.build_scenarios_client(
1811
+ capabilities, transport, channel_state
1812
+ )
1813
+
1814
+ adapter = OutboundAdapter(
1815
+ capabilities,
1816
+ events_spool=events_spool,
1817
+ events_client=events_client,
1818
+ results_client=results_client,
1819
+ artifacts_client=artifacts_client,
1820
+ channel_state=channel_state,
1821
+ extra_secret_values=(
1822
+ *deps.peek_secret_values(),
1823
+ *tuple(simulator_secret_values.values()),
1824
+ ),
1825
+ clock=deps.clock,
1826
+ flush_window_seconds=deps.flush_window_seconds,
1827
+ )
1828
+
1829
+ cancel_state = CancelState(deps.cancel_path)
1830
+ restore_sigterm = deps.install_sigterm_handler(cancel_state)
1831
+ # held outside the try so an exception on any path after this line still lets the
1832
+ # `finally` below close whatever was actually provisioned.
1833
+ pool: WorldPool | None = None
1834
+ call_runner: CallRunner | None = None
1835
+
1836
+ def cancel_requested() -> bool:
1837
+ requested = cancel_state.requested() or adapter.is_fenced
1838
+ if requested:
1839
+ # "120s from the cancel signal / TTL / terminal event" -- whichever comes first;
1840
+ # a cancel/fence observed here starts the clock even though the terminal event (which
1841
+ # also arms it, unconditionally) may not land until much later.
1842
+ adapter.arm_flush_window()
1843
+ return requested
1844
+
1845
+ async def _bounded_close() -> None:
1846
+ if pool is None:
1847
+ return
1848
+ remaining = adapter.deadline()
1849
+ if remaining is None:
1850
+ await pool.close()
1851
+ return
1852
+ timeout = max(0.0, remaining - time.monotonic())
1853
+ try:
1854
+ await asyncio.wait_for(pool.close(), timeout=timeout)
1855
+ except asyncio.TimeoutError:
1856
+ logger.warning(
1857
+ "pool.close() did not finish within the remaining flush window (%.1fs); "
1858
+ "WorldPool already latches itself closed on entry to close(), so the top-level "
1859
+ "finally's own pool.close() call cannot retry the teardown -- the provisioner may "
1860
+ "be left not fully torn down until close()'s latch ordering changes",
1861
+ timeout,
1862
+ )
1863
+
1864
+ async def _finish(
1865
+ stage: HarnessStage,
1866
+ *,
1867
+ reason: ob.TerminalReason | None = None,
1868
+ failure: dict[str, Any] | None = None,
1869
+ complete: bool,
1870
+ scheduler_result: tuple[HostedScheduler, RunResult] | None = None,
1871
+ ) -> int:
1872
+ """Terminal event -> drain -> bounded close, in that order, for every path that
1873
+ reaches a genuine terminal stage -- FAILED (via `_fail`; this now covers the pre-run
1874
+ failure branches too, not just post-run ones), CANCELED, an aborted RunResult, COMPLETED.
1875
+ Spending close()'s W-engine teardown time BEFORE a single terminal event is queued is
1876
+ exactly the inversion this ordering guards against.
1877
+
1878
+ `scheduler_result` is only ever passed by the three call sites reached AFTER
1879
+ `scheduler.run()` -- pre-run terminals (`_fail`, the boundary `_canceled()` checks) have
1880
+ no `RunResult` and pass nothing, so this stays a no-op there."""
1881
+ # Artifact bytes must be uploaded before the terminal-referenced complete manifest. The
1882
+ # terminal event itself remains before receipts and the manifest on the outbound channel.
1883
+ await adapter.ensure_terminal_artifacts(
1884
+ work_directory=work_directory,
1885
+ stage=stage,
1886
+ failure=failure,
1887
+ )
1888
+ await adapter.emit_terminal(stage=stage, reason=reason, failure=failure)
1889
+ # (outbound-channels.md v1.3 Sequencing): skipped receipts go out AFTER the terminal
1890
+ # event, never before -- placed here so no return path below can skip this call while
1891
+ # still delivering the terminal. A fenced result emits nothing further (the scheduler's
1892
+ # own no-op covers it too; checked here as well so a fenced run never even attempts it).
1893
+ # Best-effort like every other post-terminal emission in this module: a failure here must
1894
+ # not undo the terminal already spooled above or change the exit code below.
1895
+ if scheduler_result is not None:
1896
+ finished_scheduler, run_result = scheduler_result
1897
+ if run_result.fenced is None:
1898
+ try:
1899
+ # `emit_terminal` above only spools the terminal locally -- flushed to the
1900
+ # wire here, BEFORE the skipped-receipt pushes below, so a receipt's own
1901
+ # trailing flush can never deliver the terminal as a side effect and land it
1902
+ # after that receipt on the wire. Not itself wrapped in the wait_for below --
1903
+ # it already threads the same deadline through every retry it makes, and it
1904
+ # runs first, so its own delivery attempt is never the thing a timeout cuts off.
1905
+ await adapter.flush_terminal(deadline=adapter.deadline())
1906
+ if not adapter.is_fenced:
1907
+ # `emit_skipped_receipts`/`receipt()` have no deadline plumbing of their
1908
+ # own (`push()`/`aflush_events()` run with `deadline=None`) -- a
1909
+ # degraded-but-alive events channel can retry every skipped scenario's
1910
+ # receipt for the full `RetryPolicy` budget, scaling with how many
1911
+ # scenarios were cut short and blowing past the flush window the gateway
1912
+ # tears the sandbox down at. Bounded the same way `_bounded_close` bounds
1913
+ # `pool.close()`: past the deadline, stop trying and fall through to close.
1914
+ remaining = adapter.deadline()
1915
+ timeout = (
1916
+ None
1917
+ if remaining is None
1918
+ else max(0.0, remaining - time.monotonic())
1919
+ )
1920
+ try:
1921
+ await asyncio.wait_for(
1922
+ finished_scheduler.emit_skipped_receipts(run_result),
1923
+ timeout=timeout,
1924
+ )
1925
+ except asyncio.TimeoutError:
1926
+ logger.warning(
1927
+ "flush window exhausted before emit_skipped_receipts finished; "
1928
+ "remaining scenarios' receipts were not sent"
1929
+ )
1930
+ except Exception as exc: # noqa: BLE001 - post-terminal telemetry, never fatal
1931
+ logger.error("emit_skipped_receipts failed: %s", exc)
1932
+ # unlinked AFTER the terminal event, not before -- every terminal path shares the
1933
+ # "secrets are no longer needed past this point" rule, but an unlink failure (a read-only
1934
+ # or non-owned /run/futureagi) must never cost the one event that proves the job reached a
1935
+ # terminal state at all. missing_ok=True still no-ops on paths where the provider's own
1936
+ # §4.4 close() already removed the file; a genuine OSError is logged, not raised --
1937
+ # deleting an already-unneeded file is best-effort, not load-bearing.
1938
+ try:
1939
+ deps.secrets_path.unlink(missing_ok=True)
1940
+ except OSError as exc:
1941
+ logger.warning("secrets.json unlink failed: %s", exc)
1942
+ if adapter.is_fenced:
1943
+ await _bounded_close()
1944
+ return EXIT_FENCED
1945
+ # the exit code comes from drain()'s own post-hoc fence check (deadline computed AFTER
1946
+ # emit_terminal, which is what arms the flush window), never a stale pre-drain read.
1947
+ fenced = await adapter.drain(deadline=adapter.deadline(), complete=complete)
1948
+ await _bounded_close()
1949
+ if fenced:
1950
+ return EXIT_FENCED
1951
+ # §0.6 v1.14: the terminal was decided but the final drain could not flush it (the events
1952
+ # channel failed) or the platform permanently rejected the terminal item itself -- exit 0
1953
+ # would claim a flush that provably never happened and silently lose the run's evidence.
1954
+ if adapter.terminal_undelivered:
1955
+ return EXIT_TERMINAL_UNDELIVERED
1956
+ return EXIT_OK
1957
+
1958
+ async def _canceled(
1959
+ *,
1960
+ scheduler_result: tuple[HostedScheduler, RunResult] | None = None,
1961
+ ) -> int:
1962
+ return await _finish(
1963
+ HarnessStage.CANCELED,
1964
+ reason=cancel_state.reason(),
1965
+ complete=False,
1966
+ scheduler_result=scheduler_result,
1967
+ )
1968
+
1969
+ async def _fail(
1970
+ *, domain: FailureDomain, fail_stage: HarnessStage, code: str, message: str
1971
+ ) -> int:
1972
+ # routed through `_finish` -- terminal event first, pool close (bounded) after, for
1973
+ # every pre-run failure branch too, not just the post-run ones `_finish` already covered.
1974
+ return await _finish(
1975
+ HarnessStage.FAILED,
1976
+ failure={
1977
+ "domain": domain.value,
1978
+ "stage": fail_stage.value,
1979
+ "code": code,
1980
+ "message": message,
1981
+ },
1982
+ complete=True,
1983
+ )
1984
+
1985
+ try:
1986
+ # 1 (cont'd). job.json (§0.2).
1987
+ try:
1988
+ job = load_job(job_path)
1989
+ except Exception as exc: # noqa: BLE001 - a malformed job.json has no typed error to catch
1990
+ # EXIT_CRASHED (not a FAILED terminal) even though a channel now exists -- a
1991
+ # malformed job.json means `job.seed`/`job.agent`/etc are not trustworthy enough to
1992
+ # build a reportable failure from, and every downstream stage assumes a valid `job`.
1993
+ logger.error("job.json invalid: %s", exc)
1994
+ return EXIT_CRASHED
1995
+
1996
+ observability.begin(
1997
+ job.job_id, job.run_id, (job.metadata or {}).get("telemetry")
1998
+ )
1999
+
2000
+ if job.seed is None:
2001
+ logger.warning(
2002
+ "job.seed is null; spine §1 guarantees a concrete integer -- using 0"
2003
+ )
2004
+ job_seed = job.seed if job.seed is not None else 0
2005
+ parallelism = resolve_parallelism(job)
2006
+ secret_purposes = job_secret_purposes(job)
2007
+ # `ProcessRuntimeProvider` deletes `secrets.json` on its FIRST `provision()` call, inside
2008
+ # `pool.start()` below -- this capture must happen (and does: `job` is only just now
2009
+ # available, but `pool.start()` is still ~50 lines further down) strictly BEFORE that
2010
+ # point, same constraint `deps.peek_secret_values()` already satisfies for redaction,
2011
+ # above at adapter construction. Alias-preserving so `CallRunnerImpl` can pick e.g.
2012
+ # `LIVEKIT_API_KEY` out of the map by name.
2013
+ target_provider_secret_values = deps.peek_target_provider_secret_values(
2014
+ secret_purposes
2015
+ )
2016
+ # Platform simulator credentials arrive through simulator-secrets.json, not the
2017
+ # customer-controlled secrets.json. Preserve that separately loaded channel all the way
2018
+ # into CallRunnerContext. Any legacy simulator-purpose refs are merged first so the
2019
+ # platform channel wins on alias collisions and cannot be overridden by a submitted job.
2020
+ simulator_provider_secret_values = {
2021
+ **deps.peek_simulator_provider_secret_values(secret_purposes),
2022
+ **simulator_secret_values,
2023
+ }
2024
+ adapter.configure_artifacts(
2025
+ job.artifacts
2026
+ ) # level table + budget, now that job.json is known.
2027
+
2028
+ adapter.stage_changed(HarnessStage.VALIDATING_ENVIRONMENT)
2029
+ await adapter.aflush_events()
2030
+
2031
+ # Bundle authoring is not this module's (see the class docstrings above) -- injected.
2032
+ try:
2033
+ manifest, bundle_dir = await asyncio.to_thread(
2034
+ deps.bundle_source.load,
2035
+ job,
2036
+ source=source,
2037
+ work_directory=work_directory,
2038
+ )
2039
+ except BundleUnavailableError as exc:
2040
+ return await _fail(
2041
+ domain=FailureDomain.ENVIRONMENT,
2042
+ fail_stage=HarnessStage.VALIDATING_ENVIRONMENT,
2043
+ code=exc.code,
2044
+ message=exc.message,
2045
+ )
2046
+
2047
+ # 2. Preflight -- BEFORE any provision (§2e). `parallelism` is the RAW requested value
2048
+ # (never clamped), so an out-of-1..8 W fails HERE with `parallelism_out_of_range`, per
2049
+ # §2e.7, rather than being silently laundered into a valid one.
2050
+ try:
2051
+ await asyncio.to_thread(
2052
+ preflight_bundle,
2053
+ bundle_dir,
2054
+ manifest,
2055
+ parallelism=parallelism,
2056
+ secret_refs=secret_purposes,
2057
+ )
2058
+ except PreflightError as exc:
2059
+ return await _fail(
2060
+ domain=FailureDomain.ENVIRONMENT,
2061
+ fail_stage=HarnessStage.VALIDATING_ENVIRONMENT,
2062
+ code=exc.code,
2063
+ message=exc.message,
2064
+ )
2065
+
2066
+ # cancel/fence check at the post-preflight stage boundary.
2067
+ if cancel_requested():
2068
+ return await _canceled()
2069
+
2070
+ # 4/5. Provision -- ProcessRuntimeProvider, hosted lane never passes
2071
+ # require_declared_user=False (the provider defaults it True on its own; the local lane's
2072
+ # opt-out is a construction-site concern, not this module's). §4.5b's provider mutex is
2073
+ # `WorldPool`'s own `_provider_lock` now (mutation-verified: it serializes
2074
+ # provision/reset/close/healthy under one lock) -- wired directly, no extra wrapper.
2075
+ provider = deps.build_provider(capabilities, transport)
2076
+ pool = WorldPool(
2077
+ provider,
2078
+ bundle=manifest,
2079
+ source=source,
2080
+ bundle_dir=bundle_dir,
2081
+ work_directory=work_directory,
2082
+ instances=parallelism,
2083
+ outbound=adapter,
2084
+ )
2085
+ try:
2086
+ await pool.start()
2087
+ except (ob.HostedFencedError, ob.HostedAttemptSupersededError):
2088
+ # defensive -- nothing today routes a channel error through `pool.start()`, but a
2089
+ # fenced attempt must never fall into the bare `Exception` handler below and get a
2090
+ # terminal FAILED event synthesized for it.
2091
+ await _bounded_close()
2092
+ return EXIT_FENCED
2093
+ except ob.HostedChannelFailedError as exc:
2094
+ # `_fail` closes the pool itself now, AFTER the terminal event (via `_finish`) --
2095
+ # closing here first was the same close-before-terminal inversion that the terminal -> drain -> close ordering fixes elsewhere.
2096
+ return await _fail(
2097
+ domain=FailureDomain.PLATFORM_SYNC,
2098
+ fail_stage=HarnessStage.VALIDATING_SCENARIOS,
2099
+ code="scenario_preallocation_failed",
2100
+ message=str(exc),
2101
+ )
2102
+ except ProcessRuntimeError as exc:
2103
+ # §2f's own CARRIED domain (never the flattened `infrastructure`/"provision_failed"
2104
+ # every provisioning failure used to get), stage `building_environment` per §2f.
2105
+ if adapter.is_fenced:
2106
+ await _bounded_close()
2107
+ return EXIT_FENCED
2108
+ return await _fail(
2109
+ domain=_process_runtime_error_domain(exc),
2110
+ fail_stage=HarnessStage.BUILDING_ENVIRONMENT,
2111
+ code=_section_2f_code(exc.code),
2112
+ message=str(exc),
2113
+ )
2114
+ except Exception as exc: # noqa: BLE001 - genuinely untyped -> infrastructure is the honest default
2115
+ if adapter.is_fenced:
2116
+ await _bounded_close()
2117
+ return EXIT_FENCED
2118
+ return await _fail(
2119
+ domain=FailureDomain.INFRASTRUCTURE,
2120
+ fail_stage=HarnessStage.BUILDING_ENVIRONMENT,
2121
+ code=_section_2f_code("provision_failed"),
2122
+ message=f"provision_failed: {exc}",
2123
+ )
2124
+
2125
+ # baseline_frozen + parallelism_degraded from build.json. The whole
2126
+ # block is guarded -- a malformed build.json value must degrade to a `log`, never kill a
2127
+ # run that has already provisioned real worlds.
2128
+ degrade_emitted = False
2129
+ try:
2130
+ build_output = await asyncio.to_thread(load_build_output, work_directory)
2131
+ except WorldFactoryError:
2132
+ build_output = {}
2133
+ try:
2134
+ for store in build_output.get("stores", []):
2135
+ if store.get("baseline_reference"):
2136
+ adapter.baseline_frozen(
2137
+ inputs_digest=str(store.get("inputs_digest", "")),
2138
+ baseline_ref=str(store.get("baseline_reference", "")),
2139
+ )
2140
+ degrade_reason = build_output.get("degrade_reason")
2141
+ if degrade_reason:
2142
+ requested = int(
2143
+ build_output.get("requested_parallelism") or parallelism
2144
+ )
2145
+ effective = int(build_output.get("effective_parallelism") or 1)
2146
+ # `ParallelismDegradedPayload` requires `1 <= effective < requested` --
2147
+ # `fixed_port` is recorded at `instances == 1` too (provider-side gap), where
2148
+ # `effective == requested == 1` is not representable as a degrade at all.
2149
+ if effective < requested:
2150
+ adapter.parallelism_degraded(
2151
+ requested=requested,
2152
+ effective=effective,
2153
+ reason=str(degrade_reason),
2154
+ )
2155
+ degrade_emitted = True
2156
+ else:
2157
+ await adapter.log(
2158
+ level="warning",
2159
+ message=(
2160
+ f"degrade recorded ({degrade_reason}) with requested==effective=="
2161
+ f"{requested}; no parallelism_degraded event is representable"
2162
+ ),
2163
+ )
2164
+ except Exception as exc: # noqa: BLE001 - malformed build.json must never crash a live run
2165
+ await adapter.log(
2166
+ level="warning",
2167
+ message=f"build.json degrade/baseline block malformed: {exc}",
2168
+ )
2169
+ # `pool.effective_size` is the ground truth for how many worlds actually exist --
2170
+ # if it's short of what was requested and build.json's own `degrade_reason` didn't already
2171
+ # announce it (a runtime degrade build.json doesn't record), say so loudly rather
2172
+ # than silently.
2173
+ if not degrade_emitted and pool.effective_size < parallelism:
2174
+ await adapter.log(
2175
+ level="warning",
2176
+ message=(
2177
+ f"world pool effective_size={pool.effective_size} < requested "
2178
+ f"parallelism={parallelism}, but build.json recorded no representable "
2179
+ "degrade_reason"
2180
+ ),
2181
+ )
2182
+ await adapter.aflush_events()
2183
+
2184
+ # cancel/fence check at the post-provision stage boundary.
2185
+ if cancel_requested():
2186
+ return await _canceled()
2187
+
2188
+ # 3. Scenario pre-allocation (spine §5 step 3.5). Generation is not this module's; the
2189
+ # pre-allocation CLIENT (ScenariosClient) is.
2190
+ adapter.stage_changed(HarnessStage.VALIDATING_SCENARIOS)
2191
+ await adapter.aflush_events()
2192
+ world_factory = deps.build_world_factory(work_directory)
2193
+ # An injected `ScenarioSource` (every test, every future caller) always wins -- the real
2194
+ # bundle-reading adapter (scenario_source.py) only steps in when the default
2195
+ # `NotWiredScenarioSource` is still in place AND the bundle actually carries a `scenarios/`
2196
+ # directory (the LAYOUT DECISION's presence test). A bundle without one keeps the existing
2197
+ # typed `ScenarioSourceNotWired` failure below -- no regression for a job whose scenarios
2198
+ # are not generated yet.
2199
+ scenario_source = deps.scenario_source
2200
+ if isinstance(scenario_source, NotWiredScenarioSource) and bundle_has_scenarios(
2201
+ bundle_dir
2202
+ ):
2203
+ scenario_source = BundleScenarioSource()
2204
+ try:
2205
+ scenarios = await scenario_source.build(
2206
+ job,
2207
+ manifest,
2208
+ scenarios_client,
2209
+ pool=pool,
2210
+ world_factory=world_factory,
2211
+ bundle_dir=bundle_dir,
2212
+ )
2213
+ except (ob.HostedFencedError, ob.HostedAttemptSupersededError):
2214
+ # `ScenariosClient._post` re-raises these after latching `channel_state` -- a fence
2215
+ # here must exit 3 with no terminal event, never fall through to the generic handler.
2216
+ await _bounded_close()
2217
+ return EXIT_FENCED
2218
+ except ob.HostedChannelFailedError as exc:
2219
+ # `ScenariosClient._post` has already latched `channel_state` by the time this branch
2220
+ # runs -- `emit_terminal` still spools the terminal locally (it never touches the
2221
+ # network), but the drain that would flush it inherits the same latched channel and can
2222
+ # never deliver. `_finish` detects exactly this (the terminal's own spool sequence never
2223
+ # gets acked) and reports it honestly rather than claiming a flush that cannot happen.
2224
+ return await _fail(
2225
+ domain=FailureDomain.PLATFORM_SYNC,
2226
+ fail_stage=HarnessStage.VALIDATING_SCENARIOS,
2227
+ code="scenario_preallocation_failed",
2228
+ message=str(exc),
2229
+ )
2230
+ except (ScenarioSourceNotWired, ScenarioPreallocationError) as exc:
2231
+ if adapter.is_fenced:
2232
+ await _bounded_close()
2233
+ return EXIT_FENCED
2234
+ return await _fail(
2235
+ domain=FailureDomain.PLATFORM_SYNC,
2236
+ fail_stage=HarnessStage.VALIDATING_SCENARIOS,
2237
+ code="scenario_preallocation_failed",
2238
+ message=str(exc),
2239
+ )
2240
+ except ScenarioDocumentInvalid as exc:
2241
+ # A scenario document that will not even compile is a generation-stage content defect
2242
+ # (deterministic on retry), never a transport failure -- same rationale as
2243
+ # `_SCENARIO_ENTRY_INVALID_CODE`'s other use below, reused rather than inventing a new
2244
+ # code for the same pair of (domain, stage).
2245
+ if adapter.is_fenced:
2246
+ await _bounded_close()
2247
+ return EXIT_FENCED
2248
+ return await _fail(
2249
+ domain=FailureDomain.ENVIRONMENT,
2250
+ fail_stage=HarnessStage.VALIDATING_SCENARIOS,
2251
+ code=_SCENARIO_ENTRY_INVALID_CODE,
2252
+ message=str(exc),
2253
+ )
2254
+
2255
+ # Defense against a malformed scenario entry (K1) reaching the scheduler, which reads
2256
+ # `scenario_key`/`sub_goals`/`setup`/`ready` with plain attribute access and would raise
2257
+ # AttributeError instead of failing the job cleanly.
2258
+ scenario_defect = _validate_scenarios(scenarios)
2259
+ if scenario_defect is not None:
2260
+ if adapter.is_fenced:
2261
+ await _bounded_close()
2262
+ return EXIT_FENCED
2263
+ return await _fail(
2264
+ domain=FailureDomain.ENVIRONMENT,
2265
+ fail_stage=HarnessStage.VALIDATING_SCENARIOS,
2266
+ code=_SCENARIO_ENTRY_INVALID_CODE,
2267
+ message=scenario_defect,
2268
+ )
2269
+
2270
+ # cancel/fence check at the post-pre-allocation stage boundary.
2271
+ if cancel_requested():
2272
+ return await _canceled()
2273
+
2274
+ # 5/6. Scheduler wiring.
2275
+ adapter.stage_changed(HarnessStage.RUNNING)
2276
+ await adapter.aflush_events()
2277
+ call_runner_context = CallRunnerContext(
2278
+ job=job,
2279
+ bundle_dir=bundle_dir,
2280
+ work_directory=work_directory,
2281
+ evidence_seam=manifest.runtime.evidence_seam,
2282
+ target_provider_secret_values=target_provider_secret_values,
2283
+ simulator_provider_secret_values=simulator_provider_secret_values,
2284
+ attempt_number=capabilities.attempt_number,
2285
+ source_directory=source,
2286
+ )
2287
+ call_runner = deps.build_call_runner(adapter, call_runner_context)
2288
+ scheduler = HostedScheduler(
2289
+ pool=pool,
2290
+ world_factory=world_factory,
2291
+ call_runner=call_runner,
2292
+ outbound=adapter,
2293
+ job_seed=job_seed,
2294
+ cancel_requested=cancel_requested,
2295
+ )
2296
+ result: RunResult = await scheduler.run(scenarios)
2297
+
2298
+ # 7. Terminal + exit codes. Terminal -> drain -> close (bounded), never close() first.
2299
+ if cancel_state.requested():
2300
+ return await _canceled(scheduler_result=(scheduler, result))
2301
+ if result.aborted is not None:
2302
+ # v1.14 §5.4 pass-through: `result.aborted.domain`/`.code` carry straight through, not
2303
+ # flattened to a fixed infrastructure/world_pool_exhausted pair -- the scheduler already
2304
+ # resolves whether every world failed on the SAME never-retried §2f code (environment
2305
+ # or agent domain) or a mixed set (`world_pool_exhausted`/`infrastructure`), and this
2306
+ # just reports that verdict unchanged.
2307
+ return await _finish(
2308
+ HarnessStage.FAILED,
2309
+ failure={
2310
+ "domain": result.aborted.domain,
2311
+ "stage": HarnessStage.RUNNING.value,
2312
+ "code": result.aborted.code,
2313
+ "message": result.aborted.message,
2314
+ },
2315
+ # The scheduler has emitted the errored receipt plus synthesized skipped
2316
+ # receipts for every scenario before reaching this branch. "complete" is an
2317
+ # evidence-delivery property, not a success flag: only cancellation may submit
2318
+ # an intentionally partial manifest.
2319
+ complete=True,
2320
+ scheduler_result=(scheduler, result),
2321
+ )
2322
+ # `complete: true` only on a genuine, nothing-cut-short COMPLETED terminal.
2323
+ return await _finish(
2324
+ HarnessStage.COMPLETED, complete=True, scheduler_result=(scheduler, result)
2325
+ )
2326
+ finally:
2327
+ if pool is not None:
2328
+ try:
2329
+ await (
2330
+ pool.close()
2331
+ ) # idempotent backstop for any path above that missed one.
2332
+ except Exception: # noqa: BLE001 - a finally must never mask the real exit path
2333
+ logger.exception("pool.close() failed in the run_job finally backstop")
2334
+ if call_runner is not None:
2335
+ close_call_runner = getattr(call_runner, "close", None)
2336
+ if callable(close_call_runner):
2337
+ try:
2338
+ result = close_call_runner()
2339
+ if hasattr(result, "__await__"):
2340
+ await result
2341
+ except Exception: # noqa: BLE001 - cleanup must never mask the real exit path
2342
+ logger.exception("call runner close failed in the run_job finally backstop")
2343
+ restore_sigterm()
2344
+
2345
+
2346
+ # =================================================================================================
2347
+ # CLI -- spine §0 step 5's frozen invocation line:
2348
+ # `python -m fi.alk.harness.hosted_entrypoint /work/job.json --source /work/source --output
2349
+ # /work/artifacts`
2350
+ # =================================================================================================
2351
+
2352
+
2353
+ def main(argv: list[str] | None = None) -> int:
2354
+ parser = argparse.ArgumentParser(prog="alk-harness-worker")
2355
+ parser.add_argument("job", type=Path, help="typed HarnessJob JSON (/work/job.json)")
2356
+ parser.add_argument(
2357
+ "--source", required=True, type=Path, help="/work/source checkout root"
2358
+ )
2359
+ parser.add_argument("--output", required=True, type=Path, help="/work/artifacts")
2360
+ args = parser.parse_args(argv)
2361
+ try:
2362
+ return asyncio.run(run_job(args.job, args.source, args.output))
2363
+ finally:
2364
+ observability.end()
2365
+
2366
+
2367
+ if __name__ == "__main__":
2368
+ raise SystemExit(main())
2369
+
2370
+
2371
+ __all__ = [
2372
+ "CANCEL_SIGNAL_PATH",
2373
+ "EXIT_BOOT_FAILURE",
2374
+ "EXIT_CRASHED",
2375
+ "EXIT_FENCED",
2376
+ "EXIT_OK",
2377
+ "EXIT_TERMINAL_UNDELIVERED",
2378
+ "BundleSource",
2379
+ "BundleUnavailableError",
2380
+ "CallRunnerNotWired",
2381
+ "CancelState",
2382
+ "DefaultBundleSource",
2383
+ "HostedEntrypointDeps",
2384
+ "NotWiredCallRunner",
2385
+ "NotWiredScenarioSource",
2386
+ "OutboundAdapter",
2387
+ "ProcessWorldFactory",
2388
+ "ScenarioPreallocationError",
2389
+ "ScenarioSource",
2390
+ "ScenarioSourceNotWired",
2391
+ "ScenariosClient",
2392
+ "WorldFactoryError",
2393
+ "install_sigterm_handler",
2394
+ "job_secret_purposes",
2395
+ "load_build_output",
2396
+ "load_job",
2397
+ "main",
2398
+ "peek_secret_values",
2399
+ "resolve_parallelism",
2400
+ "row_counts_for_capability",
2401
+ "run_job",
2402
+ ]