agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,3252 @@
1
+ """Outbound reporting — `outbound-channels.md` v1.3, paired with `hosted-execution-seams.md` v1.11
2
+ (the spine). All guest -> platform traffic (events, result receipts, artifacts) is outbound HTTPS;
3
+ the platform never calls in. Two halves:
4
+
5
+ Foundations (part 1): the platform capability declaration the gateway uploads to
6
+ `/run/futureagi/capabilities.json`, the byte-exact canonical serialization every digest in the
7
+ contract is built from, and the durable local spool (with its monotonic sequence allocator) that
8
+ emission sits behind so a killed process within a live sandbox never loses or duplicates a
9
+ record. (A killed sandbox is deleted by the gateway, spool and all -- there is no in-sandbox
10
+ restart producer in the spine today, so the cross-restart recovery machinery guards a scenario
11
+ that isn't triggerable yet, but the in-process failed-write and watermark-durability guarantees
12
+ are load-bearing from the first event.)
13
+
14
+ Transport (part 2): a channel-neutral `Transport` protocol plus a `requests`-backed production
15
+ implementation; the closed status-code error map (`classify_response`) shared by all three channel
16
+ clients; and the clients themselves — `EventsClient` (batches spooled events, advances the spool
17
+ watermark only on confirmed delivery), `ResultsClient` (typed `ResultReceiptDraft` + delivery), and
18
+ `ArtifactsClient` (content-addressed upload + `ArtifactManifestDraft` + delivery), each sharing one
19
+ retry/backoff engine (`_perform_with_retry`) that raises on the two channel-ending outcomes
20
+ (`HostedFencedError` for 401/403, `HostedChannelFailedError` for 404 exhausted) and returns a typed
21
+ `ChannelError` for everything else a caller must log-and-continue on.
22
+
23
+ `HostedEvent`/`HostedEventDraft` model Channel 1's wire shape (the "hosted event model" the spine's
24
+ implementation-delta list calls for — `sequence`/`attempt_id`/`attempt_number`/`stage`/`digest` —
25
+ distinct from `fi.simulate.runtime.events.CanonicalEvent`, which remains the local-SDK wire and is
26
+ untouched by this module).
27
+
28
+ Redaction (v1.3 Channel 1; seams v1.11 §3): `redact_outbound_text` scrubs URL userinfo
29
+ (`scheme://user:pw@` -> `scheme://user:***@`) plus an adapter-supplied secret-value list, applied
30
+ inside `build_event_record`/`build_result_receipt` to every free-text field the contract names
31
+ (`log.message`, `world_unhealthy.cause`, `terminal`/receipt `failure.message`, sub_goal/evaluation
32
+ `reason`). It is NOT the full "same secret-content scan as the artifact sealer" the contract also
33
+ requires — that broader scan is a separate, sealer-side obligation this module does not implement.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import hashlib
39
+ import json
40
+ import logging
41
+ import os
42
+ import random
43
+ import re
44
+ import time
45
+ import uuid
46
+ from collections.abc import Callable, Collection, Iterator, Mapping
47
+ from dataclasses import dataclass, field
48
+ from datetime import datetime, timezone
49
+ from enum import Enum
50
+ from pathlib import Path
51
+ from threading import RLock
52
+ from typing import Annotated, Any, ClassVar, Literal, Protocol
53
+ from urllib.parse import urlparse
54
+
55
+ try: # N30: fcntl is POSIX-only; the guest is Linux-only, but the module must still IMPORT
56
+ import fcntl
57
+ except ImportError: # pragma: no cover - non-POSIX
58
+ fcntl = None # type: ignore[assignment]
59
+
60
+ import requests
61
+ from pydantic import (
62
+ AfterValidator,
63
+ BaseModel,
64
+ ConfigDict,
65
+ Field,
66
+ JsonValue,
67
+ ValidationError,
68
+ field_serializer,
69
+ model_validator,
70
+ )
71
+
72
+ from .job import FailureDomain, HarnessStage
73
+
74
+ logger = logging.getLogger(__name__)
75
+
76
+ CAPABILITIES_SCHEMA_VERSION = "futureagi.harness-capabilities.v1"
77
+ CAPABILITIES_PATH = "/run/futureagi/capabilities.json"
78
+ EVENT_SCHEMA_VERSION = "futureagi.harness-event.v1"
79
+ RESULT_SCHEMA_VERSION = "futureagi.harness-result.v1"
80
+ MANIFEST_SCHEMA_VERSION = "futureagi.harness-manifest.v1"
81
+
82
+ # "Channel 1" limits (outbound-channels.md v1.3) that every producer of an event record needs
83
+ # to honor before it ever reaches a transport client. MIN-12: consumed by EventsClient.flush()
84
+ # (stamps `schema_version`, clamps the batch to EVENTS_MAX_BATCH) and by HostedEventDraft's own
85
+ # size check -- no longer just declared and unused.
86
+ EVENTS_MAX_BATCH = 100
87
+ EVENT_PAYLOAD_MAX_BYTES = 32 * 1024
88
+ # N7: a cumulative-bytes cap on top of EVENTS_MAX_BATCH's event-count cap. 100 events * 32KB could
89
+ # reach ~3.2MB; this keeps a proactively-built batch comfortably under a common ~1MB ingress cap
90
+ # (nginx's default) so 413 is the exception, not the steady state -- EventsClient.flush() still
91
+ # halves and retries reactively on an observed 413 regardless of this cap.
92
+ EVENTS_MAX_BATCH_BYTES = 900_000
93
+ # §3a: uploads over this size use chunked transfer; consumed by ArtifactsClient's default
94
+ # chunk_threshold_bytes.
95
+ ARTIFACT_CHUNKED_UPLOAD_THRESHOLD_BYTES = 64 * 1024 * 1024
96
+ # "Sequencing"/"Flush window": the drain deadline from the cancel signal, TTL, or terminal event.
97
+ # N5/N24: this module does not compute a deadline from it -- every public client method
98
+ # (EventsClient.flush / ResultsClient.push / ArtifactsClient.upload / .push_manifest) instead
99
+ # accepts an explicit `deadline: float | None` (a `time.monotonic()` value). The adapter (P10) is
100
+ # the one process-wide owner of "when did the window start," so it is the one that turns this
101
+ # constant into the deadline value it passes in -- not this module.
102
+ FLUSH_WINDOW_SECONDS = 120
103
+
104
+
105
+ # =================================================================================================
106
+ # Canonicalization -- the byte-exact serialization every digest in the contract is built from.
107
+ # =================================================================================================
108
+
109
+
110
+ class OutboundError(RuntimeError):
111
+ """Generic typed failure for this module's canonicalization/digest layer -- same `code`/
112
+ `message` shape as `CapabilitiesError`/`OutboundSpoolError`, used where neither of those is the
113
+ right domain (canonicalization itself, and shape checks that run before any spool or
114
+ capabilities object exists)."""
115
+
116
+ def __init__(self, code: str, message: str) -> None:
117
+ self.code = code
118
+ self.message = message
119
+ super().__init__(f"{code}: {message}")
120
+
121
+
122
+ def canonical_bytes(value: Any) -> bytes:
123
+ """The contract's canonical form ("Canonicalization (every digest in this file)"):
124
+ ``json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False)``,
125
+ encoded UTF-8 (v1.3 pins `allow_nan=False` explicitly). This is the ONLY place that call is
126
+ made -- every digest function in this module goes through it, so a change to the algorithm
127
+ cannot happen in only one of them.
128
+
129
+ `allow_nan=False` changes nothing about the bytes for any value that was already valid JSON --
130
+ NaN/Infinity are not RFC 8259, so a float that would have silently produced unparseable bytes
131
+ now fails loudly here instead of downstream at the platform's parser.
132
+
133
+ Never re-derive an already-spooled record's bytes by calling this again on retry: a dict's key
134
+ order is stable within one process but nothing guarantees float formatting or dict construction
135
+ order is bit-identical across a restart. `OutboundSpool` hands back the literal bytes it wrote;
136
+ those are what a retry re-sends, per the contract's "serialize once, spool the bytes, re-send
137
+ verbatim; never re-serialize on retry."
138
+ """
139
+ try:
140
+ return json.dumps(
141
+ value,
142
+ sort_keys=True,
143
+ separators=(",", ":"),
144
+ ensure_ascii=False,
145
+ allow_nan=False,
146
+ ).encode("utf-8")
147
+ except ValueError as exc:
148
+ raise OutboundError("canonical_value_not_finite", str(exc)) from exc
149
+
150
+
151
+ def sha256_digest(data: bytes) -> str:
152
+ return "sha256:" + hashlib.sha256(data).hexdigest()
153
+
154
+
155
+ def event_payload_digest(payload: dict[str, Any]) -> str:
156
+ """Event digest scope: the `payload` object alone (not the envelope around it)."""
157
+ return sha256_digest(canonical_bytes(payload))
158
+
159
+
160
+ def _json_native_offense(value: Any, path: str) -> str | None:
161
+ """Walks `value` looking for the first thing `canonical_bytes` cannot represent for a reason
162
+ OTHER than NaN/Infinity (which `canonical_bytes` itself catches): a non-JSON-native Python
163
+ value (`datetime`, `Decimal`, `UUID`, ...) or a non-string dict key. Returns the offending key
164
+ path (e.g. `"call.started_at"`), or `None` if the tree is clean."""
165
+ if value is None or isinstance(value, (bool, int, float, str)):
166
+ return None
167
+ if isinstance(value, dict):
168
+ for key, item in value.items():
169
+ if not isinstance(key, str):
170
+ return (
171
+ f"{path}[<non-string-key>:{key!r}]"
172
+ if path
173
+ else f"<non-string-key>:{key!r}"
174
+ )
175
+ offense = _json_native_offense(item, f"{path}.{key}" if path else key)
176
+ if offense is not None:
177
+ return offense
178
+ return None
179
+ if isinstance(value, list):
180
+ for index, item in enumerate(value):
181
+ offense = _json_native_offense(item, f"{path}[{index}]")
182
+ if offense is not None:
183
+ return offense
184
+ return None
185
+ return path or "<root>"
186
+
187
+
188
+ def whole_object_digest(obj: dict[str, Any]) -> str:
189
+ """Receipt/manifest digest scope: the whole object with the `digest` key ABSENT.
190
+
191
+ The key is popped, never set to `None` -- the contract is explicit that "absent and null are
192
+ different bytes," so silently keeping `digest: null` in the canonicalized form would compute a
193
+ different (wrong) hash than what the platform verifies against.
194
+
195
+ Unlike an event payload (already pydantic-validated as `dict[str, JsonValue]` before it ever
196
+ reaches `event_payload_digest`), receipts and manifests reach this function as hand-built
197
+ dicts from whatever calls it -- a bare `TypeError` from `json.dumps` on a non-JSON-native value
198
+ or a non-string key is a debugging dead end with no indication of WHERE in the object the bad
199
+ value lives. This walks the tree first and raises a typed `OutboundError` naming the offending
200
+ key path instead.
201
+ """
202
+ core = {key: value for key, value in obj.items() if key != "digest"}
203
+ offense = _json_native_offense(core, "")
204
+ if offense is not None:
205
+ raise OutboundError(
206
+ "digest_value_not_json_native", f"non-JSON-native value at: {offense}"
207
+ )
208
+ return sha256_digest(canonical_bytes(core))
209
+
210
+
211
+ _DIGEST_PATTERN = re.compile(r"sha256:[0-9a-f]{64}")
212
+
213
+
214
+ def is_valid_digest(value: str) -> bool:
215
+ return bool(_DIGEST_PATTERN.fullmatch(value))
216
+
217
+
218
+ _RFC3339_MILLIS_PATTERN = re.compile(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z")
219
+
220
+
221
+ def format_rfc3339_millis(value: datetime) -> str:
222
+ """The contract's exact timestamp wire form ("Timestamps: RFC 3339, UTC, `Z`, millisecond
223
+ precision"). Unlike `HostedEvent.emitted_at` (an envelope field the event digest scope never
224
+ covers, so its exact string form doesn't matter), receipt/manifest timestamps such as
225
+ `call.started_at` sit INSIDE the `whole_object_digest` scope -- what we hash must be byte-
226
+ identical to what we send, so those fields are plain `str` on the wire models, produced only
227
+ through this function, never through a datetime's default serialization (which pydantic would
228
+ render as e.g. `+00:00` offset and six-digit microseconds, not `Z` and milliseconds).
229
+ """
230
+ if value.tzinfo is None:
231
+ raise ValueError("naive_datetime_not_allowed")
232
+ utc = value.astimezone(timezone.utc)
233
+ return utc.strftime("%Y-%m-%dT%H:%M:%S.") + f"{utc.microsecond // 1000:03d}Z"
234
+
235
+
236
+ def is_valid_rfc3339_millis(value: str) -> bool:
237
+ return bool(_RFC3339_MILLIS_PATTERN.fullmatch(value))
238
+
239
+
240
+ def _require_utc_millis(value: datetime) -> datetime:
241
+ """Shared `AfterValidator` for every `datetime` field this module serializes through
242
+ `format_rfc3339_millis` (`HostedEventDraft.emitted_at`, `HostedCapabilities.expires_at`):
243
+ rejects a naive datetime outright (the contract's wire form has no naive representation),
244
+ converts any other offset to UTC, and truncates to millisecond precision so the VALUE itself --
245
+ not just its string rendering -- matches what gets sent on the wire."""
246
+ if value.tzinfo is None:
247
+ raise ValueError("naive_datetime_not_allowed")
248
+ utc = value.astimezone(timezone.utc)
249
+ return utc.replace(microsecond=(utc.microsecond // 1000) * 1000)
250
+
251
+
252
+ UtcMillisDatetime = Annotated[datetime, AfterValidator(_require_utc_millis)]
253
+
254
+
255
+ # N9/P3: URL userinfo -- `scheme://user:pw@host` -> `scheme://user:***@host`, matching the seams
256
+ # contract's own example (`postgresql://harness:***@...`). The username group is now OPTIONAL
257
+ # (`redis://:pw@host`, the canonical empty-username shape for Redis/RabbitMQ/Mongo, is a real
258
+ # managed-store DSN form) and so is the whole password group (`https://<token>@host`, the standard
259
+ # way a bearer token appears in git/registry output) -- a bare userinfo token is masked outright
260
+ # rather than left verbatim on the theory that "a username alone is not a secret," which is false
261
+ # for a token.
262
+ _USERINFO_PATTERN = re.compile(
263
+ r"([a-zA-Z][a-zA-Z0-9+.\-]*://)([^\s:/?#@]*)(:[^\s/?#]*)?@"
264
+ )
265
+
266
+
267
+ def _mask_userinfo(match: re.Match[str]) -> str:
268
+ scheme, user, password = match.group(1), match.group(2), match.group(3)
269
+ return f"{scheme}{user}:***@" if password is not None else f"{scheme}***@"
270
+
271
+
272
+ def redact_outbound_text(value: str, extra_secret_values: tuple[str, ...] = ()) -> str:
273
+ """Scrubs a single free-text field before it can leave the sandbox on any of the three
274
+ channels (outbound-channels.md v1.3 "Redaction (enforced before emit)"; hosted-execution-
275
+ seams.md v1.11 §3 "any outbound projection ... redacts userinfo"). Two things, applied in
276
+ order:
277
+
278
+ 1. URL userinfo (`_USERINFO_PATTERN`/`_mask_userinfo`, above) -- a password is masked and the
279
+ username kept (matching the contract's own `postgresql://harness:***@...` example); a bare
280
+ token/username-only userinfo (no `:`) is masked outright, since that shape is how a bearer
281
+ token appears, not a username.
282
+ 2. `extra_secret_values` -- exact-substring replacement for a caller-supplied list of secret
283
+ values. Always `()` today: the adapter (P10) is what will know the job's declared secrets
284
+ and pass them in -- this parameter exists now so `build_event_record`/`build_result_receipt`
285
+ never need to change shape when that wiring lands.
286
+
287
+ NOT a general secret-content scanner -- the contract's "same secret-content scan as the
288
+ artifact sealer" is a separate, sealer-side obligation. This is the narrow subset this module
289
+ can enforce on every string field it controls without false-positiving on ordinary diagnostic
290
+ text.
291
+ """
292
+ redacted = _USERINFO_PATTERN.sub(_mask_userinfo, value)
293
+ for secret in extra_secret_values:
294
+ if secret:
295
+ redacted = redacted.replace(secret, "***")
296
+ return redacted
297
+
298
+
299
+ # =================================================================================================
300
+ # Capabilities -- `/run/futureagi/capabilities.json`, "Authentication" section.
301
+ # =================================================================================================
302
+
303
+
304
+ class CapabilitiesError(RuntimeError):
305
+ """A capability declaration is missing, malformed, or fails a shape rule.
306
+
307
+ Mirrors the `code`/`message` shape `BundleV2Error`/`PreflightError` use elsewhere in this
308
+ package. `code` is one of the nine closed values in outbound-channels.md v1.3's "Capabilities-
309
+ file rejection table" -- see `load_capabilities`, which is the sole place that maps a raw
310
+ failure onto one of them.
311
+ """
312
+
313
+ def __init__(self, code: str, message: str) -> None:
314
+ self.code = code
315
+ self.message = message
316
+ super().__init__(f"{code}: {message}")
317
+
318
+
319
+ class HostedEndpoints(BaseModel):
320
+ """The four outbound routes, per attempt. Shape-only checks (trailing slash, https) live here
321
+ as defense-in-depth for direct construction; `load_capabilities` runs the SAME checks earlier,
322
+ outside pydantic, so each has its own `CapabilitiesError.code` instead of collapsing into the
323
+ generic `capabilities_field_invalid` (see v1.3's rejection table)."""
324
+
325
+ model_config = ConfigDict(extra="forbid")
326
+
327
+ events: str
328
+ results: str
329
+ artifacts: str
330
+ scenarios: str
331
+ ingress: str | None = None
332
+
333
+ @model_validator(mode="after")
334
+ def _shape(self) -> "HostedEndpoints":
335
+ for name in ("events", "results", "artifacts", "scenarios", "ingress"):
336
+ value = getattr(self, name)
337
+ if value is None:
338
+ continue
339
+ if not value or not value.endswith("/"):
340
+ raise ValueError(f"{name} endpoint must end with '/'")
341
+ if not value.startswith("https://"):
342
+ raise ValueError(f"{name} endpoint must use https")
343
+ return self
344
+
345
+
346
+ class HostedCapabilities(BaseModel):
347
+ """The per-attempt bearer plus the four endpoint URLs. Loaded once at emitter startup from the
348
+ file the gateway uploads (§0 step 4) -- see `load_capabilities`, which also implements the
349
+ contract's "loaded into memory ... and unlinked" lifetime rule."""
350
+
351
+ model_config = ConfigDict(extra="forbid")
352
+
353
+ schema_version: str
354
+ job_id: str = Field(min_length=1)
355
+ attempt_id: str = Field(min_length=1)
356
+ attempt_number: int = Field(ge=1)
357
+ fence: str = Field(min_length=1)
358
+ expires_at: UtcMillisDatetime
359
+ token: str = Field(min_length=1)
360
+ endpoints: HostedEndpoints
361
+
362
+ @model_validator(mode="after")
363
+ def _schema_shape(self) -> "HostedCapabilities":
364
+ # Defense-in-depth only -- `load_capabilities` checks this first, outside pydantic, so it
365
+ # can raise `capabilities_schema_unsupported` specifically rather than the generic
366
+ # `capabilities_field_invalid` this validator's `ValueError` would collapse into.
367
+ if self.schema_version != CAPABILITIES_SCHEMA_VERSION:
368
+ raise ValueError(f"unsupported schema_version: {self.schema_version}")
369
+ return self
370
+
371
+ @field_serializer("expires_at")
372
+ def _serialize_expires_at(self, value: datetime) -> str:
373
+ return format_rfc3339_millis(value)
374
+
375
+ def auth_headers(self) -> dict[str, str]:
376
+ """ "Every request: `Authorization: Bearer <token>` + `X-Harness-Fence: <fence>`." Pure
377
+ formatting -- issuing the request itself is a P8 transport-client concern."""
378
+ return {"Authorization": f"Bearer {self.token}", "X-Harness-Fence": self.fence}
379
+
380
+ def event_builder(
381
+ self, *, extra_secret_values: tuple[str, ...] = ()
382
+ ) -> Callable[..., dict[str, Any]]:
383
+ """A `build_event_record`-shaped callable with `job_id`/`attempt_id`/`attempt_number`
384
+ closed over from THIS capabilities object. The contract's `403 attempt_mismatch` fires when
385
+ an event's identity disagrees with the token authenticating it -- binding these three
386
+ fields here makes that class of caller bug unrepresentable at the call site instead of a
387
+ runtime 403 discovered mid-attempt.
388
+
389
+ P4: `extra_secret_values` is bound here too, alongside identity, rather than left as a
390
+ per-call parameter -- `build_event_record` could not previously receive the job's declared
391
+ secret list at all through this binder, so a caller wired for `event_builder()` had no way
392
+ to satisfy N9's redaction requirement on `log.message`/`terminal.failure.message` without
393
+ routing around this method entirely. Binding once here matches how identity is already
394
+ bound and gives the adapter one place to get it wrong instead of two.
395
+ """
396
+
397
+ def build(
398
+ *,
399
+ event_id: str,
400
+ emitted_at: datetime,
401
+ stage: HarnessStage,
402
+ type: OutboundEventType,
403
+ payload: dict[str, JsonValue],
404
+ ) -> dict[str, Any]:
405
+ return build_event_record(
406
+ event_id=event_id,
407
+ job_id=self.job_id,
408
+ attempt_id=self.attempt_id,
409
+ attempt_number=self.attempt_number,
410
+ emitted_at=emitted_at,
411
+ stage=stage,
412
+ type=type,
413
+ payload=payload,
414
+ extra_secret_values=extra_secret_values,
415
+ )
416
+
417
+ return build
418
+
419
+
420
+ def _endpoint_matches_attempt(url: str, attempt_id: str) -> bool:
421
+ """ "an endpoint's `<attempt_id>` path segment disagrees with the declared `attempt_id`"
422
+ (`capabilities_attempt_mismatch`, v1.3) -- checked as a whole path segment, not a substring, so
423
+ an attempt_id that happens to be a substring of another segment can't produce a false match."""
424
+ segments = [segment for segment in urlparse(url).path.split("/") if segment]
425
+ return attempt_id in segments
426
+
427
+
428
+ def _redact_validation_error(exc: ValidationError) -> str:
429
+ """Builds a `capabilities_field_invalid` message from `loc`/`msg` only -- pydantic's default
430
+ `str(exc)` embeds each failing field's `input_value`, and the capabilities file carries the
431
+ bearer token; a caller that logs this message must never be able to leak it."""
432
+ parts = []
433
+ for error in exc.errors():
434
+ loc = ".".join(str(part) for part in error.get("loc", ()))
435
+ msg = error.get("msg", "")
436
+ parts.append(f"{loc}: {msg}" if loc else msg)
437
+ return "; ".join(parts) or "capabilities file failed validation"
438
+
439
+
440
+ def _warn_if_capabilities_file_insecure(target: Path) -> None:
441
+ """ "owner svc-control, mode 0600" is the contract's posture for this file, but a wrong mode or
442
+ owner is NOT a load-time rejection (MIN-5, fail-safe): the bearer is only a per-attempt token
443
+ that expires on its own, and refusing to load it entirely over a permissions mistake would turn
444
+ a minor hardening gap into a hard attempt failure. Loud warning only."""
445
+ try:
446
+ info = target.stat()
447
+ except OSError:
448
+ return
449
+ mode = info.st_mode & 0o777
450
+ if mode != 0o600:
451
+ logger.warning(
452
+ "%s: capabilities file mode is %o, expected 0600 -- a world/group-readable bearer in "
453
+ "a multi-user sandbox is a leak risk (not blocking the load)",
454
+ target,
455
+ mode,
456
+ )
457
+ try:
458
+ running_uid = os.geteuid()
459
+ except AttributeError:
460
+ return # os.geteuid() is POSIX-only
461
+ if info.st_uid != running_uid:
462
+ logger.warning(
463
+ "%s: capabilities file is owned by uid %s, not the running uid %s -- expected owner "
464
+ "svc-control per the contract (not blocking the load)",
465
+ target,
466
+ info.st_uid,
467
+ running_uid,
468
+ )
469
+
470
+
471
+ def load_capabilities(
472
+ path: str | Path = CAPABILITIES_PATH,
473
+ *,
474
+ unlink: bool = True,
475
+ now: Callable[[], datetime] | None = None,
476
+ on_unlink_failure: Callable[[OSError], None] | None = None,
477
+ ) -> HostedCapabilities:
478
+ """Parse and validate one capabilities file against outbound-channels.md v1.3's closed
479
+ "Capabilities-file rejection table" (nine codes, all reachable as `CapabilitiesError.code`).
480
+
481
+ A guest that cannot load this file has no channel at all -- no token, no endpoints -- so it
482
+ cannot report its own failure; every branch below raises before any network-capable object
483
+ exists. Most checks run BEFORE `HostedCapabilities.model_validate`, outside any pydantic
484
+ validator: wrapping them in a pydantic `ValidationError` (the old shape) meant they only ever
485
+ surfaced as the generic `capabilities_field_invalid`, never as their own named code -- checking
486
+ here first makes each one an independently raised, independently testable `CapabilitiesError`.
487
+
488
+ ``unlink=True`` (the default) implements "loaded into memory at emitter startup and unlinked" --
489
+ the file is only ever removed AFTER a successful parse and validation, never before, so a
490
+ crash mid-load leaves the file in place for the next attempt to read rather than destroying the
491
+ only copy of a not-yet-consumed bearer. A failure to unlink is never fatal to an otherwise-
492
+ successful load (the sandbox is destroyed by the gateway at attempt end regardless) but is no
493
+ longer silently swallowed either: it is reported via `on_unlink_failure` if given, else logged
494
+ (MIN-7) -- the caller can still tell a 0600 bearer may be lingering on disk.
495
+
496
+ ``now`` is injectable (defaults to the real clock) so `capabilities_expired` is testable without
497
+ manipulating the wall clock.
498
+ """
499
+ target = Path(path).expanduser()
500
+ try:
501
+ exists = target.is_file()
502
+ except OSError as exc:
503
+ raise CapabilitiesError("capabilities_file_unreadable", str(exc)) from exc
504
+ if not exists:
505
+ raise CapabilitiesError("capabilities_file_missing", str(target))
506
+ _warn_if_capabilities_file_insecure(target)
507
+
508
+ try:
509
+ text = target.read_text(encoding="utf-8")
510
+ except OSError as exc:
511
+ raise CapabilitiesError("capabilities_file_unreadable", str(exc)) from exc
512
+ try:
513
+ raw = json.loads(text)
514
+ except json.JSONDecodeError as exc:
515
+ raise CapabilitiesError("capabilities_file_malformed", str(exc)) from exc
516
+ if not isinstance(raw, dict):
517
+ raise CapabilitiesError("capabilities_file_malformed", "not a JSON object")
518
+
519
+ schema_version = raw.get("schema_version")
520
+ if schema_version != CAPABILITIES_SCHEMA_VERSION:
521
+ raise CapabilitiesError("capabilities_schema_unsupported", str(schema_version))
522
+
523
+ attempt_id = raw.get("attempt_id")
524
+ endpoints_raw = raw.get("endpoints")
525
+ if isinstance(endpoints_raw, dict):
526
+ for name in ("events", "results", "artifacts", "scenarios"):
527
+ value = endpoints_raw.get(name)
528
+ if not isinstance(value, str):
529
+ continue # missing/wrong-typed -- a shape error pydantic below will catch
530
+ if not value.endswith("/"):
531
+ raise CapabilitiesError(
532
+ "capabilities_endpoint_invalid",
533
+ f"endpoints.{name} must end with '/'",
534
+ )
535
+ if not value.startswith("https://"):
536
+ raise CapabilitiesError(
537
+ "capabilities_endpoint_insecure", f"endpoints.{name} must use https"
538
+ )
539
+ if (
540
+ isinstance(attempt_id, str)
541
+ and attempt_id
542
+ and not _endpoint_matches_attempt(value, attempt_id)
543
+ ):
544
+ raise CapabilitiesError(
545
+ "capabilities_attempt_mismatch",
546
+ f"endpoints.{name} does not carry the declared attempt_id {attempt_id!r}",
547
+ )
548
+
549
+ try:
550
+ capabilities = HostedCapabilities.model_validate(raw)
551
+ except ValidationError as exc:
552
+ raise CapabilitiesError(
553
+ "capabilities_field_invalid", _redact_validation_error(exc)
554
+ ) from exc
555
+
556
+ current_time = (now or (lambda: datetime.now(timezone.utc)))()
557
+ if capabilities.expires_at <= current_time:
558
+ raise CapabilitiesError(
559
+ "capabilities_expired", f"expires_at={capabilities.expires_at.isoformat()}"
560
+ )
561
+
562
+ if unlink:
563
+ try:
564
+ target.unlink()
565
+ except OSError as exc:
566
+ if on_unlink_failure is not None:
567
+ on_unlink_failure(exc)
568
+ else:
569
+ logger.warning(
570
+ "%s: failed to unlink the capabilities file after a successful load (%s) -- a "
571
+ "0600 bearer may still be on disk; the sandbox is destroyed at attempt end "
572
+ "regardless, so this does not fail the load",
573
+ target,
574
+ exc,
575
+ )
576
+ return capabilities
577
+
578
+
579
+ # =================================================================================================
580
+ # Channel 1 -- Events. The closed `type` vocabulary and each type's payload shape.
581
+ # =================================================================================================
582
+
583
+
584
+ class DegradeReason(str, Enum):
585
+ CONFORMANCE_GATE_FAILED = "conformance_gate_failed"
586
+ FIXED_PORT = "fixed_port"
587
+
588
+
589
+ class LogLevel(str, Enum):
590
+ DEBUG = "debug"
591
+ INFO = "info"
592
+ WARNING = "warning"
593
+ ERROR = "error"
594
+
595
+
596
+ class TerminalReason(str, Enum):
597
+ TTL_EXCEEDED = "ttl_exceeded"
598
+ USER_CANCELED = "user_canceled"
599
+
600
+
601
+ class OutboundEventType(str, Enum):
602
+ STAGE_CHANGED = "stage_changed"
603
+ PARALLELISM_DEGRADED = "parallelism_degraded"
604
+ BASELINE_FROZEN = "baseline_frozen"
605
+ BASELINE_INPUTS_CHANGED = "baseline_inputs_changed"
606
+ WORLD_UNHEALTHY = "world_unhealthy"
607
+ SCENARIO_STARTED = "scenario_started"
608
+ SCENARIO_RETRIED = "scenario_retried"
609
+ LOG = "log"
610
+ TERMINAL = "terminal"
611
+
612
+
613
+ class StageChangedPayload(BaseModel):
614
+ """No `populate_by_name` -- the wire key is `from` (a Python keyword, hence the `from_stage`
615
+ attribute name + alias), and this model is validation-only (`HostedEventDraft` never
616
+ normalizes `self.payload`; it emits the caller's dict verbatim). Allowing population by the
617
+ attribute name too would let a caller who writes `from_stage` in their payload dict pass
618
+ validation while spooling an undefined wire key -- `populate_by_name=True` previously made
619
+ exactly that mistake succeed silently."""
620
+
621
+ model_config = ConfigDict(extra="forbid")
622
+
623
+ from_stage: HarnessStage | None = Field(alias="from")
624
+ to: HarnessStage
625
+
626
+
627
+ class ParallelismDegradedPayload(BaseModel):
628
+ model_config = ConfigDict(extra="forbid")
629
+
630
+ requested: int = Field(ge=1)
631
+ effective: int = Field(ge=1)
632
+ reason: DegradeReason
633
+
634
+ @model_validator(mode="after")
635
+ def _range(self) -> "ParallelismDegradedPayload":
636
+ if not (1 <= self.effective < self.requested):
637
+ raise ValueError(
638
+ f"parallelism_degraded_effective_out_of_range: effective={self.effective} "
639
+ f"requested={self.requested}"
640
+ )
641
+ return self
642
+
643
+
644
+ class BaselineFrozenPayload(BaseModel):
645
+ model_config = ConfigDict(extra="forbid")
646
+
647
+ inputs_digest: str
648
+ baseline_ref: str
649
+
650
+
651
+ class BaselineInputsChangedPayload(BaseModel):
652
+ model_config = ConfigDict(extra="forbid")
653
+
654
+ previous_digest: str | None
655
+ current_digest: str
656
+
657
+
658
+ class WorldUnhealthyPayload(BaseModel):
659
+ model_config = ConfigDict(extra="forbid")
660
+
661
+ world_index: int = Field(ge=0)
662
+ cause: str = Field(max_length=200)
663
+
664
+
665
+ class ScenarioStartedPayload(BaseModel):
666
+ model_config = ConfigDict(extra="forbid")
667
+
668
+ scenario_key: str = Field(min_length=1)
669
+ world_index: int = Field(ge=0)
670
+ scenario_attempt: Literal[1, 2]
671
+
672
+
673
+ class ScenarioRetriedPayload(BaseModel):
674
+ model_config = ConfigDict(extra="forbid")
675
+
676
+ scenario_key: str = Field(min_length=1)
677
+ from_world: int = Field(ge=0)
678
+ to_world: int = Field(ge=0)
679
+
680
+
681
+ class LogPayload(BaseModel):
682
+ model_config = ConfigDict(extra="forbid")
683
+
684
+ level: LogLevel
685
+ message: str
686
+
687
+
688
+ _LOG_TRUNCATION_MARKER = "…[truncated]"
689
+
690
+
691
+ def truncate_log_message(
692
+ level: str, message: str, *, max_payload_bytes: int = EVENT_PAYLOAD_MAX_BYTES
693
+ ) -> str:
694
+ """The `log` event's own contract rule -- "truncated to fit with a trailing `…[truncated]`
695
+ marker" -- unlike the other eight event types, which are hard-rejected when oversized (M8:
696
+ `log` is the contract's designated escape hatch for reporting every other permanent failure, so
697
+ the one channel meant to report an oversized diagnostic must not itself throw on size).
698
+
699
+ Sizing is against `canonical_bytes({"level": level, "message": <candidate>})`, the exact bytes
700
+ `HostedEventDraft`'s own size check measures, so a truncated message is guaranteed to fit
701
+ before it ever reaches that check. A no-op when already within budget.
702
+ """
703
+ if len(canonical_bytes({"level": level, "message": message})) <= max_payload_bytes:
704
+ return message
705
+ if (
706
+ len(canonical_bytes({"level": level, "message": _LOG_TRUNCATION_MARKER}))
707
+ > max_payload_bytes
708
+ ):
709
+ raise OutboundError(
710
+ "log_payload_budget_too_small",
711
+ f"max_payload_bytes={max_payload_bytes} cannot fit even the truncation marker",
712
+ )
713
+ lo, hi, best = 0, len(message), ""
714
+ while lo <= hi:
715
+ mid = (lo + hi) // 2
716
+ candidate = message[:mid] + _LOG_TRUNCATION_MARKER
717
+ if (
718
+ len(canonical_bytes({"level": level, "message": candidate}))
719
+ <= max_payload_bytes
720
+ ):
721
+ best = candidate
722
+ lo = mid + 1
723
+ else:
724
+ hi = mid - 1
725
+ return best
726
+
727
+
728
+ class TerminalFailure(BaseModel):
729
+ """The terminal event's `failure` shape: `{domain, stage, code, message}` -- a leaner subset of
730
+ `job.HarnessFailure` (no `retryable`/`details`), matching §"Event `type` vocabulary" exactly so
731
+ a canonicalized terminal payload never carries fields the contract doesn't name."""
732
+
733
+ model_config = ConfigDict(extra="forbid")
734
+
735
+ domain: FailureDomain
736
+ stage: HarnessStage
737
+ code: str
738
+ message: str
739
+
740
+
741
+ class ScenarioCounts(BaseModel):
742
+ model_config = ConfigDict(extra="forbid")
743
+
744
+ passed: int = Field(ge=0)
745
+ failed: int = Field(ge=0)
746
+ errored: int = Field(ge=0)
747
+ skipped: int = Field(ge=0)
748
+
749
+
750
+ class TerminalPayload(BaseModel):
751
+ model_config = ConfigDict(extra="forbid")
752
+
753
+ stage: HarnessStage
754
+ reason: TerminalReason | None
755
+ failure: TerminalFailure | None
756
+ scenario_counts: ScenarioCounts
757
+
758
+ @model_validator(mode="after")
759
+ def _terminal_stage(self) -> "TerminalPayload":
760
+ if not self.stage.terminal:
761
+ raise ValueError(f"terminal_event_stage_not_terminal: {self.stage.value}")
762
+ return self
763
+
764
+
765
+ _PAYLOAD_MODELS: dict[OutboundEventType, type[BaseModel]] = {
766
+ OutboundEventType.STAGE_CHANGED: StageChangedPayload,
767
+ OutboundEventType.PARALLELISM_DEGRADED: ParallelismDegradedPayload,
768
+ OutboundEventType.BASELINE_FROZEN: BaselineFrozenPayload,
769
+ OutboundEventType.BASELINE_INPUTS_CHANGED: BaselineInputsChangedPayload,
770
+ OutboundEventType.WORLD_UNHEALTHY: WorldUnhealthyPayload,
771
+ OutboundEventType.SCENARIO_STARTED: ScenarioStartedPayload,
772
+ OutboundEventType.SCENARIO_RETRIED: ScenarioRetriedPayload,
773
+ OutboundEventType.LOG: LogPayload,
774
+ OutboundEventType.TERMINAL: TerminalPayload,
775
+ }
776
+
777
+
778
+ class HostedEventDraft(BaseModel):
779
+ """A Channel 1 event before spool-assigned `sequence`. The caller supplies `digest` itself
780
+ (computed via `event_payload_digest`) -- the model then re-derives it and rejects a mismatch,
781
+ so a caller can never accidentally spool a record whose embedded digest disagrees with its own
782
+ payload bytes.
783
+ """
784
+
785
+ model_config = ConfigDict(extra="forbid")
786
+
787
+ event_id: str = Field(min_length=1, max_length=64)
788
+ job_id: str = Field(min_length=1)
789
+ attempt_id: str = Field(min_length=1)
790
+ attempt_number: int = Field(ge=1)
791
+ emitted_at: UtcMillisDatetime
792
+ stage: HarnessStage
793
+ type: OutboundEventType
794
+ payload: dict[str, JsonValue]
795
+ digest: str
796
+
797
+ @field_serializer("emitted_at")
798
+ def _serialize_emitted_at(self, value: datetime) -> str:
799
+ return format_rfc3339_millis(value)
800
+
801
+ @model_validator(mode="after")
802
+ def _validate(self) -> "HostedEventDraft":
803
+ # `max_length=64` above counts characters; the contract says "opaque <=64 chars," but the
804
+ # platform's column is presumably bytes -- a multi-byte-UTF-8 id could pass the character
805
+ # count and still overflow it (N-2).
806
+ if len(self.event_id.encode("utf-8")) > 64:
807
+ raise ValueError(f"event_id_too_long_in_bytes: {self.event_id!r}")
808
+ if not is_valid_digest(self.digest):
809
+ raise ValueError(f"event_digest_invalid: {self.digest!r}")
810
+ expected = event_payload_digest(self.payload)
811
+ if self.digest != expected:
812
+ raise ValueError("event_digest_mismatch")
813
+ if len(canonical_bytes(self.payload)) > EVENT_PAYLOAD_MAX_BYTES:
814
+ raise ValueError(f"event_payload_too_large: {self.event_id}")
815
+
816
+ model_cls = _PAYLOAD_MODELS[self.type]
817
+ try:
818
+ model_cls.model_validate(self.payload)
819
+ except ValidationError as exc:
820
+ raise ValueError(
821
+ f"event_payload_invalid: {self.type.value}: {exc}"
822
+ ) from exc
823
+
824
+ if (
825
+ self.type is OutboundEventType.STAGE_CHANGED
826
+ and self.payload.get("to") != self.stage.value
827
+ ):
828
+ raise ValueError(
829
+ "event_stage_mismatch: stage_changed.to must equal the event's stage"
830
+ )
831
+ if (
832
+ self.type is OutboundEventType.TERMINAL
833
+ and self.payload.get("stage") != self.stage.value
834
+ ):
835
+ raise ValueError(
836
+ "event_stage_mismatch: terminal.stage must equal the event's stage"
837
+ )
838
+ return self
839
+
840
+
841
+ class HostedEvent(HostedEventDraft):
842
+ """The full Channel 1 wire object, `sequence` included -- what actually gets spooled and sent.
843
+ Distinct from `fi.simulate.runtime.events.CanonicalEvent` (the untouched local-SDK wire)."""
844
+
845
+ sequence: int = Field(ge=1)
846
+
847
+
848
+ def build_event_record(
849
+ *,
850
+ event_id: str,
851
+ job_id: str,
852
+ attempt_id: str,
853
+ attempt_number: int,
854
+ emitted_at: datetime,
855
+ stage: HarnessStage,
856
+ type: OutboundEventType,
857
+ payload: dict[str, JsonValue],
858
+ extra_secret_values: tuple[str, ...] = (),
859
+ ) -> dict[str, Any]:
860
+ """Validate one event's shape and compute its digest, returning a plain dict with no
861
+ `sequence` key -- ready for `OutboundSpool.append`, which assigns `sequence` and performs the
862
+ one-time serialization. Raises `ValueError` (via pydantic) on any shape violation; callers that
863
+ want a typed/coded failure should catch `pydantic.ValidationError` themselves, matching how the
864
+ rest of this package surfaces model-layer rejections (`bundle_v2.py`, `job.py`).
865
+
866
+ N9: `redact_outbound_text` runs on every free-text field the contract names BEFORE the digest
867
+ is computed -- `log.message`, `world_unhealthy.cause`, `terminal.failure.{code,message}`,
868
+ `baseline_frozen.baseline_ref` (P8) -- so the embedded digest always matches the redacted bytes
869
+ actually spooled and sent, never the unredacted original. `log` events are then truncated to
870
+ fit (M8), also before the digest -- redact first, since truncation must size against the final
871
+ (redacted) text, not text that would still shrink again once secrets are scrubbed. Every other
872
+ event type still hard-rejects when oversized, via `HostedEventDraft`'s own size check.
873
+
874
+ P8: `failure.code` is redacted alongside `failure.message` -- both are free `str` fields (the
875
+ contract's `code` vocabularies are closed in prose, but nothing enforces that here), and
876
+ `baseline_ref` is likewise a free `str` that plausibly carries an OCI/registry reference in the
877
+ same `https://<token>@registry/...` shape `redact_outbound_text` already scrubs.
878
+ """
879
+ payload = dict(payload)
880
+ if type is OutboundEventType.LOG:
881
+ level, message = payload.get("level"), payload.get("message")
882
+ if isinstance(message, str):
883
+ message = redact_outbound_text(message, extra_secret_values)
884
+ if isinstance(level, str):
885
+ message = truncate_log_message(level, message)
886
+ payload["message"] = message
887
+ elif type is OutboundEventType.WORLD_UNHEALTHY:
888
+ cause = payload.get("cause")
889
+ if isinstance(cause, str):
890
+ payload["cause"] = redact_outbound_text(cause, extra_secret_values)
891
+ elif type is OutboundEventType.BASELINE_FROZEN:
892
+ baseline_ref = payload.get("baseline_ref")
893
+ if isinstance(baseline_ref, str):
894
+ payload["baseline_ref"] = redact_outbound_text(
895
+ baseline_ref, extra_secret_values
896
+ )
897
+ elif type is OutboundEventType.TERMINAL:
898
+ failure = payload.get("failure")
899
+ if isinstance(failure, dict):
900
+ redacted_failure = dict(failure)
901
+ if isinstance(failure.get("code"), str):
902
+ redacted_failure["code"] = redact_outbound_text(
903
+ failure["code"], extra_secret_values
904
+ )
905
+ if isinstance(failure.get("message"), str):
906
+ redacted_failure["message"] = redact_outbound_text(
907
+ failure["message"], extra_secret_values
908
+ )
909
+ payload["failure"] = redacted_failure
910
+ digest = event_payload_digest(payload)
911
+ draft = HostedEventDraft(
912
+ event_id=event_id,
913
+ job_id=job_id,
914
+ attempt_id=attempt_id,
915
+ attempt_number=attempt_number,
916
+ emitted_at=emitted_at,
917
+ stage=stage,
918
+ type=type,
919
+ payload=payload,
920
+ digest=digest,
921
+ )
922
+ return draft.model_dump(mode="json")
923
+
924
+
925
+ # =================================================================================================
926
+ # Spool -- durable on-disk queue + monotonic sequence allocator.
927
+ # =================================================================================================
928
+
929
+
930
+ @dataclass(frozen=True)
931
+ class SpooledRecord:
932
+ """One durably-appended record. `body` is the EXACT canonical bytes written to disk -- a P8
933
+ transport client re-sends `body` verbatim on retry rather than re-serializing the decoded
934
+ dict, per the contract's "serialize once ... never re-serialize on retry.\""""
935
+
936
+ sequence: int | None
937
+ body: bytes
938
+
939
+ def decode(self) -> dict[str, Any]:
940
+ return json.loads(self.body.decode("utf-8"))
941
+
942
+
943
+ class OutboundSpoolError(RuntimeError):
944
+ def __init__(self, code: str, message: str) -> None:
945
+ self.code = code
946
+ self.message = message
947
+ super().__init__(f"{code}: {message}")
948
+
949
+
950
+ def _iter_complete_records(
951
+ data: bytes,
952
+ ) -> tuple[
953
+ list[
954
+ tuple[int, bytes, dict[str, Any] | list[Any] | str | int | float | bool | None]
955
+ ],
956
+ int,
957
+ int | None,
958
+ ]:
959
+ """Shared by `_recover` and `records()`/`pending_since_watermark()` -- the ONE place spool
960
+ bytes are split into records, so both ever agree on what a "complete record" is.
961
+
962
+ Splits on the literal `b"\\n"` byte (N-1: not `bytes.splitlines()`, which also treats `\\r`/
963
+ `\\r\\n` as separators `append` never writes -- canonical JSON never contains a raw newline of
964
+ any kind, so `\\n` is the only byte that can legitimately end a line).
965
+
966
+ Returns `(records_before_corruption, valid_length, corruption_offset)` (N8):
967
+ - `records_before_corruption`: every complete, successfully decoded, non-blank line UP TO the
968
+ first corrupt one (or all of them, if none is corrupt), as `(start_offset, raw_line,
969
+ decoded_value)`, in file order.
970
+ - `corruption_offset`: the byte offset of the first `\\n`-terminated line that failed to parse
971
+ as JSON -- genuine corruption (a torn write, by definition, never got its trailing `\\n`, so
972
+ this is never that) -- or `None` if no such line was found. Once found, scanning STOPS: never
973
+ renumber or trust anything past a corrupt byte (B1).
974
+ - `valid_length`: when `corruption_offset is None`, the byte offset immediately after the last
975
+ complete line -- where `_recover` truncates away a torn tail. When corruption WAS found, this
976
+ equals `corruption_offset` and callers must NOT use it to truncate -- corrupt bytes are left
977
+ on disk, never deleted (N8: "never truncates mid-file damage").
978
+ """
979
+ records: list[tuple[int, bytes, Any]] = []
980
+ start = 0
981
+ while True:
982
+ newline_index = data.find(b"\n", start)
983
+ if newline_index == -1:
984
+ return records, start, None
985
+ line = data[start:newline_index]
986
+ line_end = newline_index + 1
987
+ if line:
988
+ try:
989
+ decoded = json.loads(line.decode("utf-8"))
990
+ except ValueError:
991
+ return records, start, start
992
+ records.append((start, line, decoded))
993
+ start = line_end
994
+
995
+
996
+ class OutboundSpool:
997
+ """Durable, crash-safe local queue for one outbound record stream (events, results, or
998
+ artifact-manifest state) -- the "fsync-first local spool" the contract requires emission to sit
999
+ behind ("Emission is an async flusher over the fsync-first local spool -- it never blocks the
1000
+ call loop"). `sequenced=True` is for Channel 1 only ("Sequencing: one allocator, one lock,
1001
+ assigned at spool append, contiguous from 1" -- receipts and the manifest carry no `sequence`
1002
+ field and use their own idempotency keys instead).
1003
+
1004
+ ONE ALLOCATOR PER STREAM (M6): `OutboundSpool(root, name, ...)` is keyed on
1005
+ `(resolved_root, name)` -- a second construction for the same key, anywhere in this process,
1006
+ returns the SAME instance rather than a second independent allocator (see `__new__`); a second
1007
+ OS PROCESS pointing at the same directory fails loudly instead, via an `fcntl.flock` on
1008
+ `<name>.spool.lock` held for the life of the owning instance.
1009
+
1010
+ Recovery rule (the contract specifies the sequencing invariant -- contiguous from 1, no gaps or
1011
+ dupes across a restart -- but not the recovery mechanism; this is the FAIL-SAFE/REVERSIBLE
1012
+ choice under the stuck-decision rule, surfaced in the P7 report):
1013
+
1014
+ The next sequence number is derived by SCANNING the spool's own JSONL log at startup, never
1015
+ from an independent counter file. A separate counter file could be durably advanced in a write
1016
+ that lands, while the record it was allocated for does not (crash between the two writes),
1017
+ producing a sequence number with no corresponding record -- a permanent, undetectable gap.
1018
+ Scanning the log makes the durably-written records themselves the only source of truth. The
1019
+ scan is then reconciled against the durable watermark (M5): `next_sequence =
1020
+ max(max_sequence_in_log, watermark) + 1` -- the watermark can be AHEAD of the log (the log lost
1021
+ already-processed records, e.g. via the directory-fsync gap M2 closes) but never behind it, so
1022
+ taking the max is always safe and never skips a record that was actually spooled.
1023
+
1024
+ A torn last line -- a crash mid-write, since a single `write()` of `body + b"\\n"` is not
1025
+ guaranteed atomic by POSIX for a regular file -- is detected (the trailing bytes don't end in
1026
+ `b"\\n"`) and the file is truncated back to the end of the last complete record before any
1027
+ further append. The next append then reuses that same sequence number rather than skipping it:
1028
+ a torn write is treated as though it never happened, closing the gap instead of creating one.
1029
+ This depends on canonical JSON never containing a raw newline byte (control characters are
1030
+ always escaped by `json.dumps`), which `append` asserts on every write. A COMPLETE line that
1031
+ still fails to parse is a different, worse fault: genuine corruption degrades the stream to its
1032
+ readable prefix rather than raising (N8, `is_corrupt`) -- see `_recover`/`records()`.
1033
+
1034
+ Registration (N4): a construction is only added to `_registry` at the END of a successful
1035
+ `__init__`, under `_registry_lock` -- never a half-built instance. A failed construction (e.g.
1036
+ `mkdir` EACCES, or `_recover` finding corruption) therefore never poisons the key: it raises
1037
+ without registering anything, and the NEXT `OutboundSpool(root, name, ...)` call starts a
1038
+ completely fresh attempt rather than returning (or conflicting with) wreckage. The real mutual-
1039
+ exclusion primitive across a same-key construction race is `_acquire_process_lock`'s `flock`
1040
+ (an OS-level device, safe across threads and processes alike) -- the registry dict on top is
1041
+ only a same-process memoization cache.
1042
+ """
1043
+
1044
+ _registry: ClassVar[dict[tuple[Path, str], "OutboundSpool"]] = {}
1045
+ _registry_lock: ClassVar[RLock] = RLock()
1046
+
1047
+ def __new__(
1048
+ cls, root: str | Path, name: str, *, sequenced: bool
1049
+ ) -> "OutboundSpool":
1050
+ resolved_root = Path(root).expanduser().resolve()
1051
+ key = (resolved_root, name)
1052
+ with cls._registry_lock:
1053
+ existing = cls._registry.get(key)
1054
+ if existing is not None:
1055
+ # getattr belt-and-braces (N4): `existing` is only ever registered after a fully
1056
+ # successful __init__, so `_sequenced` should always be set -- but never trust that
1057
+ # invariant harder than a defensive read costs.
1058
+ if getattr(existing, "_sequenced", None) != sequenced:
1059
+ raise OutboundSpoolError(
1060
+ "outbound_spool_sequenced_mismatch",
1061
+ f"{name}: existing instance has sequenced={getattr(existing, '_sequenced', None)}, "
1062
+ f"requested sequenced={sequenced}",
1063
+ )
1064
+ return existing
1065
+ return super().__new__(cls)
1066
+
1067
+ def __init__(self, root: str | Path, name: str, *, sequenced: bool) -> None:
1068
+ if getattr(self, "_initialized", False):
1069
+ return
1070
+ resolved_root = Path(root).expanduser().resolve()
1071
+ key = (resolved_root, name)
1072
+ with type(self)._registry_lock:
1073
+ if getattr(self, "_initialized", False):
1074
+ return
1075
+ lock_fd: int | None = None
1076
+ try:
1077
+ self.root = resolved_root
1078
+ self.root.mkdir(parents=True, exist_ok=True)
1079
+ try:
1080
+ os.chmod(
1081
+ self.root, 0o700
1082
+ ) # MIN-10: mkdir's mode is subject to umask
1083
+ except OSError:
1084
+ pass
1085
+ self._name = name
1086
+ self._sequenced = sequenced
1087
+ self._path = self.root / f"{name}.spool.jsonl"
1088
+ self._watermark_path = self.root / f"{name}.spool.watermark.json"
1089
+ self._lock = RLock()
1090
+ self._dir_synced = False
1091
+ self._offset_by_sequence: dict[int, int] = {}
1092
+ self._next_sequence = 1 if sequenced else None
1093
+ self._poisoned = False # N12
1094
+ self._corrupt_since_offset: int | None = None # N8
1095
+ self._forked = False # N25
1096
+ self._closed = False # P2
1097
+ self._lock_fd = self._acquire_process_lock()
1098
+ lock_fd = self._lock_fd
1099
+ self._recover()
1100
+ except BaseException:
1101
+ # N4: never leave a half-built instance registered -- it was never added (below),
1102
+ # so there is nothing to evict; just release whatever this attempt itself opened.
1103
+ if lock_fd is not None:
1104
+ try:
1105
+ os.close(lock_fd)
1106
+ except OSError:
1107
+ pass
1108
+ raise
1109
+ self._initialized = True
1110
+ type(self)._registry[key] = self
1111
+
1112
+ @classmethod
1113
+ def _forget_for_tests(cls, root: str | Path, name: str) -> None:
1114
+ """Test-only escape hatch: a real process restart naturally gets a fresh, empty registry
1115
+ (a new interpreter); simulating that WITHIN one process/test needs an explicit evict so the
1116
+ next `OutboundSpool(root, name, ...)` call re-scans the on-disk log instead of returning the
1117
+ still-live cached instance. Never called from production code."""
1118
+ resolved_root = Path(root).expanduser().resolve()
1119
+ with cls._registry_lock:
1120
+ instance = cls._registry.get((resolved_root, name))
1121
+ if instance is not None:
1122
+ instance.close()
1123
+
1124
+ @classmethod
1125
+ def _clear_registry_for_tests(cls) -> None:
1126
+ """Broader sibling of `_forget_for_tests`: releases every cached instance's lock fd and
1127
+ empties the registry. Intended for an autouse test fixture so flock fds don't accumulate
1128
+ across a whole test session."""
1129
+ with cls._registry_lock:
1130
+ instances = list(cls._registry.values())
1131
+ for instance in instances:
1132
+ instance.close()
1133
+
1134
+ def close(self) -> None:
1135
+ """N26/P2: releases this instance's process lock and evicts it from the registry, so a
1136
+ later `OutboundSpool(root, name, ...)` call re-scans the on-disk log instead of reusing this
1137
+ instance. Idempotent -- safe to call more than once, or on an instance never fully
1138
+ constructed.
1139
+
1140
+ P2: also sets `_closed`, so THIS instance -- not just the registry slot -- refuses further
1141
+ mutation. Evicting the registry entry alone left the closed instance itself fully live: a
1142
+ caller still holding a reference could keep appending with no flock held (the lock fd was
1143
+ released), and a fresh `OutboundSpool(...)` call for the same key would allocate a second,
1144
+ independent `_next_sequence` -- two live allocators for one stream, each unaware of the
1145
+ other, which is exactly the M6 invariant `close()` must not itself reopen."""
1146
+ with type(self)._registry_lock:
1147
+ key = (getattr(self, "root", None), getattr(self, "_name", None))
1148
+ if type(self)._registry.get(key) is self:
1149
+ del type(self)._registry[key]
1150
+ fd = getattr(self, "_lock_fd", None)
1151
+ if fd is not None:
1152
+ try:
1153
+ os.close(fd)
1154
+ except OSError:
1155
+ pass
1156
+ self._lock_fd = None
1157
+ self._closed = True
1158
+
1159
+ def _require_writable(self) -> None:
1160
+ """P2/P7: the single gate `append`, `advance_watermark`, and `_rewrite_retaining` all call
1161
+ before touching disk -- refuses a closed, forked, or poisoned instance instead of letting it
1162
+ silently duplicate allocators, advance a parent's watermark from a forked child, or compound
1163
+ a rollback failure `_truncate_to` already flagged as unrecoverable."""
1164
+ if self._closed:
1165
+ raise OutboundSpoolError(
1166
+ "outbound_spool_closed",
1167
+ f"{self._name}: this OutboundSpool was closed; construct a new one for this stream",
1168
+ )
1169
+ if self._forked:
1170
+ raise OutboundSpoolError(
1171
+ "outbound_spool_forked",
1172
+ f"{self._name}: this OutboundSpool was constructed before a fork; construct a "
1173
+ f"new one in the child process instead of reusing this one",
1174
+ )
1175
+ if self._poisoned:
1176
+ raise OutboundSpoolError(
1177
+ "outbound_spool_poisoned",
1178
+ f"{self._name}: a prior rollback failed and left this spool in an unknown "
1179
+ f"state; it must not be mutated again",
1180
+ )
1181
+
1182
+ def _acquire_process_lock(self) -> int:
1183
+ """M6: cross-PROCESS protection (the in-process registry above only protects against a
1184
+ second Python-level instance in this same interpreter). Held for the life of this instance
1185
+ -- released implicitly when its fd closes (process exit, `close()`, or `_forget_for_tests`
1186
+ in tests)."""
1187
+ if fcntl is None: # N30
1188
+ raise OutboundSpoolError(
1189
+ "outbound_spool_platform_unsupported",
1190
+ f"{self._name}: fcntl (POSIX file locking) is unavailable on this platform",
1191
+ )
1192
+ lock_path = self.root / f"{self._name}.spool.lock"
1193
+ fd = os.open(str(lock_path), os.O_CREAT | os.O_RDWR, 0o600)
1194
+ try:
1195
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
1196
+ except OSError as exc:
1197
+ os.close(fd)
1198
+ raise OutboundSpoolError(
1199
+ "outbound_spool_locked",
1200
+ f"{self._name}: already locked by another process",
1201
+ ) from exc
1202
+ return fd
1203
+
1204
+ def _fsync_dir(self) -> None:
1205
+ fd = os.open(str(self.root), os.O_DIRECTORY)
1206
+ try:
1207
+ os.fsync(fd)
1208
+ finally:
1209
+ os.close(fd)
1210
+
1211
+ def _report_corruption(self, offset: int) -> None:
1212
+ """N8: the client-visible half of "degrade, don't wedge" -- `is_corrupt`/`corruption_offset`
1213
+ stay true/set for the life of this instance once discovered, but the loud `logger.error`
1214
+ fires only the FIRST time (repeated reads of an already-known-corrupt spool would otherwise
1215
+ spam the log every flush cycle)."""
1216
+ with self._lock:
1217
+ already_reported = self._corrupt_since_offset is not None
1218
+ self._corrupt_since_offset = offset
1219
+ if not already_reported:
1220
+ logger.error(
1221
+ "%s: spool corrupt at byte offset %d -- the file is left untouched; only records "
1222
+ "before that offset are trusted. Reading degrades to the readable prefix rather "
1223
+ "than raising -- this is reported once per process, not per read.",
1224
+ self._name,
1225
+ offset,
1226
+ )
1227
+
1228
+ @property
1229
+ def is_corrupt(self) -> bool:
1230
+ """N8: set once `_recover` or a later read finds a genuinely corrupt (not merely torn)
1231
+ record. A caller can turn this into a `log`-kind event; nothing in this class does that
1232
+ itself (no scenario/attempt context here to build one)."""
1233
+ return self._corrupt_since_offset is not None
1234
+
1235
+ @property
1236
+ def corruption_offset(self) -> int | None:
1237
+ return self._corrupt_since_offset
1238
+
1239
+ def _recover(self) -> None:
1240
+ # NOTE: the sequenced branch below must run even when the log is missing or empty -- an
1241
+ # M2-style lost file (durable watermark, vanished log) is exactly the case M5 needs to
1242
+ # reconcile against; an early return here would skip that reconciliation entirely and
1243
+ # silently reset the allocator to 1.
1244
+ records: list[tuple[int, bytes, Any]] = []
1245
+ if self._path.exists():
1246
+ data = self._path.read_bytes()
1247
+ if data:
1248
+ records, valid_length, corruption_offset = _iter_complete_records(data)
1249
+ if corruption_offset is not None:
1250
+ # N8: never truncate mid-file damage -- leave the bytes exactly as they are,
1251
+ # trust only what came before, and let construction succeed anyway (B1 already
1252
+ # forbids renumbering past it; the OTHER extreme -- raising here -- would
1253
+ # discard every future emit, including the terminal event, forever).
1254
+ self._report_corruption(corruption_offset)
1255
+ elif valid_length < len(data):
1256
+ with self._path.open("r+b") as stream:
1257
+ stream.truncate(valid_length)
1258
+ stream.flush()
1259
+ os.fsync(
1260
+ stream.fileno()
1261
+ ) # MIN-11: durable, not left as a crash window
1262
+ if self._sequenced:
1263
+ max_sequence = 0
1264
+ offsets: dict[int, int] = {}
1265
+ for offset, _line, record in records:
1266
+ if isinstance(record, dict):
1267
+ sequence = record.get("sequence")
1268
+ if isinstance(sequence, int):
1269
+ offsets[sequence] = offset
1270
+ if sequence > max_sequence:
1271
+ max_sequence = sequence
1272
+ self._offset_by_sequence = offsets
1273
+ watermark = self.watermark()
1274
+ if watermark > max_sequence:
1275
+ logger.warning(
1276
+ "%s: watermark (%s) is ahead of the highest sequence found in the spool (%s) "
1277
+ "-- the log lost records the platform already processed; seeding "
1278
+ "next_sequence from the watermark so newly allocated sequences don't collide "
1279
+ "with ones the platform already closed",
1280
+ self._name,
1281
+ watermark,
1282
+ max_sequence,
1283
+ )
1284
+ self._next_sequence = max(max_sequence, watermark) + 1
1285
+
1286
+ def _truncate_to(self, size: int) -> None:
1287
+ """B1: a true no-op on a failed append. `_next_sequence` is only advanced AFTER a
1288
+ successful write, so the retry reuses the same sequence number -- this makes sure it reuses
1289
+ clean ground too, instead of appending immediately after torn bytes with no `\\n` between
1290
+ them (which would merge into one unparseable line the rest of this class can't recover
1291
+ from)."""
1292
+ try:
1293
+ with self._path.open("r+b") as stream:
1294
+ stream.truncate(size)
1295
+ stream.flush()
1296
+ os.fsync(stream.fileno())
1297
+ except OSError as exc:
1298
+ # N12: the write may have failed before the file even existed (nothing to truncate,
1299
+ # harmless) OR the rollback itself failed on an existing torn write -- in the latter
1300
+ # case B1's "a failed append is a true no-op" no longer holds, so poison this spool
1301
+ # rather than let a future append silently merge into the torn bytes.
1302
+ self._poisoned = True
1303
+ logger.error(
1304
+ "%s: rollback of a failed append could not truncate the spool back to %d bytes "
1305
+ "(%s) -- the file may now carry torn bytes; poisoning this spool so a caller sees "
1306
+ "a typed error instead of a future append compounding the corruption",
1307
+ self._name,
1308
+ size,
1309
+ exc,
1310
+ )
1311
+
1312
+ def append(self, record: dict[str, Any]) -> SpooledRecord:
1313
+ with self._lock:
1314
+ self._require_writable()
1315
+ if self._sequenced:
1316
+ assigned = self._next_sequence
1317
+ record = {**record, "sequence": assigned}
1318
+ elif "sequence" in record:
1319
+ raise OutboundSpoolError(
1320
+ "outbound_spool_caller_supplied_sequence",
1321
+ f"{self._name} spool does not assign sequence numbers; caller must not pass one",
1322
+ )
1323
+ body = canonical_bytes(record)
1324
+ if b"\n" in body:
1325
+ # Framing invariant `_iter_complete_records`/`_recover` depend on: canonical JSON
1326
+ # never contains a raw newline (json.dumps escapes control characters inside
1327
+ # strings), so this would only fire on a value this module's own canonicalization
1328
+ # contract disallows.
1329
+ raise OutboundSpoolError(
1330
+ "outbound_spool_record_unframable",
1331
+ f"{self._name}: record contains a raw newline",
1332
+ )
1333
+ existed_before = self._path.exists()
1334
+ size_before = self._path.stat().st_size if existed_before else 0
1335
+ try:
1336
+ with self._path.open("ab") as stream:
1337
+ stream.write(body)
1338
+ stream.write(b"\n")
1339
+ stream.flush()
1340
+ os.fsync(stream.fileno())
1341
+ except BaseException:
1342
+ self._truncate_to(size_before)
1343
+ raise
1344
+ if not existed_before and not self._dir_synced:
1345
+ # M2: the directory entry for a brand-new file isn't durable just because the
1346
+ # file's own data is -- fsync it once (not per append; existing files' entries were
1347
+ # already synced by whichever append first created them).
1348
+ self._fsync_dir()
1349
+ self._dir_synced = True
1350
+ if self._sequenced:
1351
+ self._offset_by_sequence[assigned] = size_before
1352
+ self._next_sequence = assigned + 1
1353
+ return SpooledRecord(sequence=assigned, body=body)
1354
+ return SpooledRecord(sequence=None, body=body)
1355
+
1356
+ def records(self) -> list[SpooledRecord]:
1357
+ """Reads the WHOLE log. The read itself happens OUTSIDE `self._lock` (M4): only the size
1358
+ snapshot that bounds it is taken under the lock, so the flusher's (potentially large) read
1359
+ never blocks `append`, the call loop's write path, for its duration. Safe because `append`
1360
+ only ever grows the file -- a read bounded to a size captured a moment earlier can only be
1361
+ stale, never torn. (`compact_through`/`drop` DO shrink the file and take the lock for their
1362
+ entire duration; this module assumes the flusher serializes its own reads against its own
1363
+ compactions rather than running them from two different threads.)
1364
+ """
1365
+ with self._lock:
1366
+ if not self._path.exists():
1367
+ return []
1368
+ size = self._path.stat().st_size
1369
+ with self._path.open("rb") as stream:
1370
+ data = stream.read(size)
1371
+ parsed, _valid_length, corruption_offset = _iter_complete_records(data)
1372
+ if (
1373
+ corruption_offset is not None
1374
+ ): # N8: degrade to the readable prefix, never raise here
1375
+ self._report_corruption(corruption_offset)
1376
+ out: list[SpooledRecord] = []
1377
+ for _offset, line, decoded in parsed:
1378
+ sequence = (
1379
+ decoded.get("sequence")
1380
+ if self._sequenced and isinstance(decoded, dict)
1381
+ else None
1382
+ )
1383
+ out.append(SpooledRecord(sequence=sequence, body=line))
1384
+ return out
1385
+
1386
+ def records_after(self, sequence: int) -> list[SpooledRecord]:
1387
+ if not self._sequenced:
1388
+ raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
1389
+ return [item for item in self.records() if (item.sequence or 0) > sequence]
1390
+
1391
+ @property
1392
+ def next_sequence(self) -> int:
1393
+ if not self._sequenced:
1394
+ raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
1395
+ assert self._next_sequence is not None
1396
+ return self._next_sequence
1397
+
1398
+ def watermark(self) -> int:
1399
+ """The highest-processed sequence acknowledged so far ("the watermark is highest-processed
1400
+ -- accepted AND rejected sequences both advance it"). Durable across a restart via a
1401
+ fsync'd-temp-then-rename-then-fsync'd-directory write (M1) -- a corrupt or unreadable
1402
+ watermark file DEGRADES TO 0 with a loud diagnostic rather than raising and wedging the
1403
+ spool: re-sending already-acked events is safe (at-least-once delivery + platform-side
1404
+ dedupe on `event_id`), while a permanently unreadable outbound channel is not.
1405
+ """
1406
+ if not self._sequenced:
1407
+ raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
1408
+ if not self._watermark_path.exists():
1409
+ return 0
1410
+ try:
1411
+ raw = json.loads(self._watermark_path.read_text(encoding="utf-8"))
1412
+ return int(raw["acked_through_sequence"])
1413
+ except (OSError, ValueError, KeyError, TypeError) as exc:
1414
+ logger.warning(
1415
+ "%s: watermark file is corrupt or unreadable (%s) -- degrading to 0. Re-sending "
1416
+ "already-acked events is safe (at-least-once + dedupe on event_id); wedging the "
1417
+ "spool permanently is not.",
1418
+ self._name,
1419
+ exc,
1420
+ )
1421
+ return 0
1422
+
1423
+ def advance_watermark(self, sequence: int) -> None:
1424
+ """v1.3: `acked_through_sequence` is untrusted platform input. A value outside
1425
+ `[current_watermark, next_sequence)` is rejected locally with a typed error and the
1426
+ watermark is left exactly as it was -- "a malformed ack must not be able to discard
1427
+ pending records" (M7). `sequence == current_watermark` is a legitimate no-op, not an error
1428
+ (repeating the same ack, or a same-valued out-of-order response).
1429
+
1430
+ P7: this is the one operation that DESTROYS delivery state (it durably advances what a
1431
+ future `pending_since_watermark()` will ever return again) -- a forked child or a poisoned
1432
+ instance advancing it would silently orphan every pending record below the new value, the
1433
+ N1 outcome by a different route. `_require_writable()` guards it for that reason even
1434
+ though nothing here writes to the JSONL log itself.
1435
+ """
1436
+ if not self._sequenced:
1437
+ raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
1438
+ with self._lock:
1439
+ self._require_writable()
1440
+ current = self.watermark()
1441
+ if sequence < current or sequence >= self._next_sequence:
1442
+ raise OutboundSpoolError(
1443
+ "outbound_spool_watermark_out_of_range",
1444
+ f"{self._name}: acked_through_sequence={sequence} outside the trusted range "
1445
+ f"[{current}, {self._next_sequence}) -- untrusted platform input, watermark "
1446
+ f"left unchanged",
1447
+ )
1448
+ if sequence == current:
1449
+ return
1450
+ temporary = (
1451
+ self.root
1452
+ / f"{self._name}.spool.watermark.tmp.{os.getpid()}.{uuid.uuid4().hex}"
1453
+ )
1454
+ with temporary.open("wb") as stream:
1455
+ stream.write(
1456
+ json.dumps(
1457
+ {"acked_through_sequence": sequence}, separators=(",", ":")
1458
+ ).encode("utf-8")
1459
+ )
1460
+ stream.flush()
1461
+ os.fsync(stream.fileno())
1462
+ os.replace(temporary, self._watermark_path)
1463
+ self._fsync_dir()
1464
+
1465
+ def pending_since_watermark(self) -> list[SpooledRecord]:
1466
+ """Convenience for "the guest advances its spool cursor through the watermark": every
1467
+ spooled record not yet acknowledged, in sequence order. Uses the offset `append` recorded
1468
+ for the first pending sequence to SEEK directly there (M4) instead of re-reading and
1469
+ re-parsing the whole log on every flush cycle; falls back to the generic full scan when no
1470
+ cached offset exists yet (e.g. a fresh recovery whose watermark sits past every record this
1471
+ process itself has written an offset for).
1472
+
1473
+ N13: the drained steady state (nothing pending -- what a polling flusher sees most cycles)
1474
+ is checked first and returns `[]` with NO file IO at all: `watermark + 1 == next_sequence`
1475
+ means every allocated sequence has already been acknowledged, so there is nothing on disk
1476
+ to seek to regardless of what `_offset_by_sequence` does or doesn't have cached.
1477
+ """
1478
+ watermark = self.watermark()
1479
+ with self._lock:
1480
+ if watermark + 1 == self._next_sequence:
1481
+ return []
1482
+ offset = self._offset_by_sequence.get(watermark + 1)
1483
+ if offset is None:
1484
+ return self.records_after(watermark)
1485
+ with self._lock:
1486
+ if not self._path.exists():
1487
+ return []
1488
+ size = self._path.stat().st_size
1489
+ if offset >= size:
1490
+ return []
1491
+ with self._path.open("rb") as stream:
1492
+ stream.seek(offset)
1493
+ data = stream.read(size - offset)
1494
+ parsed, _valid_length, corruption_offset = _iter_complete_records(data)
1495
+ if (
1496
+ corruption_offset is not None
1497
+ ): # N8: absolute offset -- `data` starts at `offset`
1498
+ self._report_corruption(offset + corruption_offset)
1499
+ return [
1500
+ SpooledRecord(
1501
+ sequence=decoded.get("sequence") if isinstance(decoded, dict) else None,
1502
+ body=line,
1503
+ )
1504
+ for _offset, line, decoded in parsed
1505
+ ]
1506
+
1507
+ def _rewrite_retaining(self, keep: Callable[[Any], bool]) -> None:
1508
+ """Shared by `compact_through` and `drop_many`: rewrites the log keeping only records
1509
+ `keep` accepts, via the same write-fsync/atomic-replace/fsync-directory durability shape
1510
+ `append`/`advance_watermark` use -- a crash mid-rewrite leaves either the old file intact
1511
+ or the new one complete, never a torn hybrid.
1512
+
1513
+ P1: if the log is already corrupt (N8), this REFUSES instead of rewriting. A rewrite always
1514
+ replaces the file from what it read, and reading stops at the corruption offset -- so
1515
+ rewriting on a corrupt spool would not merely skip the corrupt bytes, it would silently
1516
+ destroy every intact record PAST them too, including ones this process itself appended
1517
+ after recovery that have never been sent (the terminal event, in the worst case). That
1518
+ directly defeats N8's "the readable prefix stays usable, delivery keeps working" posture
1519
+ the moment the first drop/compact happens. Refusing costs only unbounded disk growth until
1520
+ the attempt ends (`compact_through`'s whole job) or a rejected record staying spooled but
1521
+ never re-emitted anyway, since it's at or below the watermark (`drop_many`'s whole job) --
1522
+ both strictly better than deleting undelivered records.
1523
+ """
1524
+ with self._lock:
1525
+ self._require_writable()
1526
+ if not self._path.exists():
1527
+ return
1528
+ size = self._path.stat().st_size
1529
+ with self._path.open("rb") as stream:
1530
+ data = stream.read(size)
1531
+ parsed, _valid_length, corruption_offset = _iter_complete_records(data)
1532
+ if corruption_offset is not None:
1533
+ self._report_corruption(corruption_offset)
1534
+ logger.error(
1535
+ "%s: refusing to compact/drop on a corrupt spool -- a rewrite replaces the file "
1536
+ "from what it read, and reading stops at byte %d, so every record past that "
1537
+ "offset (including not-yet-delivered ones) would be destroyed. The log is left "
1538
+ "intact and grows unbounded until the attempt ends; that is the fail-safe half "
1539
+ "of degrade-not-wedge.",
1540
+ self._name,
1541
+ corruption_offset,
1542
+ )
1543
+ return
1544
+ temporary = (
1545
+ self.root
1546
+ / f"{self._name}.spool.jsonl.tmp.{os.getpid()}.{uuid.uuid4().hex}"
1547
+ )
1548
+ offsets: dict[int, int] = {}
1549
+ offset = 0
1550
+ with temporary.open("wb") as stream:
1551
+ for _old_offset, line, decoded in parsed:
1552
+ if not keep(decoded):
1553
+ continue
1554
+ if (
1555
+ self._sequenced
1556
+ and isinstance(decoded, dict)
1557
+ and isinstance(decoded.get("sequence"), int)
1558
+ ):
1559
+ offsets[decoded["sequence"]] = offset
1560
+ stream.write(line)
1561
+ stream.write(b"\n")
1562
+ offset += len(line) + 1
1563
+ stream.flush()
1564
+ os.fsync(stream.fileno())
1565
+ os.replace(temporary, self._path)
1566
+ self._fsync_dir()
1567
+ if self._sequenced:
1568
+ self._offset_by_sequence = offsets
1569
+
1570
+ def compact_through(self, sequence: int) -> None:
1571
+ """M4: physically drops every durably-acked record (`sequence <= min(sequence,
1572
+ watermark())`) from the on-disk log, bounding its growth for a long `running` stage. The
1573
+ allocator's `next_sequence` is unaffected -- it is only ever derived from the log at
1574
+ `_recover` time, and recovery's own `max(max_sequence, watermark)` rule (M5) already
1575
+ tolerates a log whose historical records were compacted away, since none of them can be
1576
+ the true maximum (compaction only ever removes sequences at or below the watermark, and
1577
+ the watermark is always <= every pending, uncompacted sequence).
1578
+
1579
+ Clamped to the current watermark regardless of what the caller passes -- compacting past
1580
+ an event the platform hasn't actually processed yet would be irreversible data loss, and
1581
+ this module's posture throughout is fail-safe over trusting the caller.
1582
+ """
1583
+ if not self._sequenced:
1584
+ raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
1585
+ effective = min(sequence, self.watermark())
1586
+ self._rewrite_retaining(
1587
+ lambda decoded: (
1588
+ not (
1589
+ isinstance(decoded, dict)
1590
+ and isinstance(decoded.get("sequence"), int)
1591
+ and decoded["sequence"] <= effective
1592
+ )
1593
+ )
1594
+ )
1595
+
1596
+ def drop_many(self, sequences: Collection[int]) -> None:
1597
+ """The contract's rejected-event mechanism: "a rejected event is dropped from the spool ...
1598
+ it is never re-emitted" (M5). PURE physical removal (N1) -- every sequence in `sequences`
1599
+ is deleted from the on-disk log in ONE rewrite pass (N14: a batch of 100 rejections is one
1600
+ `_rewrite_retaining` call, not 100), and the watermark is left untouched.
1601
+
1602
+ N1: an earlier version had `drop` also advance the watermark to `sequence`, on the theory
1603
+ that "a rejected event closes its sequence." That let an UNTRUSTED `rejected[].sequence`
1604
+ from the platform silently orphan every pending record below it, bypassing
1605
+ `advance_watermark`'s own M7 clamp entirely -- the clamp only guards `acked_through_sequence`
1606
+ callers, and `drop` skipped straight past it. The batch-level
1607
+ `advance_watermark(acked_through_sequence)` a caller performs separately is the ONE place
1608
+ the watermark ever moves; it already covers every rejected sequence under a conformant
1609
+ platform (rejections advance the watermark by contract), and under a non-conformant one
1610
+ M7's clamp is then the single, correct chokepoint -- this method has no clamp of its own to
1611
+ bypass.
1612
+
1613
+ Writing the record's payload to the artifact spool as a `log` kind (the other half of the
1614
+ contract's drop rule) is NOT this method's job -- that hand-off needs scenario/attempt
1615
+ context and an `ArtifactsClient` this layer doesn't own; it is P9/P10 wiring, documented
1616
+ here as the seam rather than guessed at.
1617
+ """
1618
+ if not self._sequenced:
1619
+ raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
1620
+ sequence_set = set(sequences)
1621
+ if not sequence_set:
1622
+ return
1623
+ self._rewrite_retaining(
1624
+ lambda decoded: (
1625
+ not (
1626
+ isinstance(decoded, dict)
1627
+ and decoded.get("sequence") in sequence_set
1628
+ )
1629
+ )
1630
+ )
1631
+
1632
+ def drop(self, sequence: int) -> None:
1633
+ """Single-sequence convenience wrapper over `drop_many` -- see its docstring for why this
1634
+ no longer touches the watermark."""
1635
+ self.drop_many((sequence,))
1636
+
1637
+
1638
+ def _poison_after_fork() -> None:
1639
+ """N25: `os.fork()` inherits both `OutboundSpool._registry` (with a live `_next_sequence`) and
1640
+ every instance's flock fd (the SAME open file description, so the lock is merely shared, not
1641
+ contended, across parent and child) -- without this, parent and child would allocate identical
1642
+ sequence numbers with no complaint. Marks every currently-registered instance so its next
1643
+ mutating call raises instead. Not reachable via `subprocess` (fork+exec resets memory); this
1644
+ guards a bare `os.fork()` specifically."""
1645
+ with OutboundSpool._registry_lock:
1646
+ for instance in OutboundSpool._registry.values():
1647
+ instance._forked = True
1648
+
1649
+
1650
+ if hasattr(os, "register_at_fork"): # POSIX-only, like fcntl (N30)
1651
+ # P7: `before=`/`after_in_parent=` pair the registry lock around the fork itself -- without
1652
+ # this, a fork occurring while some OTHER thread holds `_registry_lock` hands the child a
1653
+ # locked RLock owned by a thread that no longer exists there, and `_poison_after_fork`'s own
1654
+ # `with OutboundSpool._registry_lock:` deadlocks at the fork point instead of poisoning
1655
+ # anything. Acquiring on `before` guarantees the FORKING thread itself owns the lock at fork
1656
+ # time, so the child's single surviving thread already owns it too -- `_poison_after_fork`'s
1657
+ # acquire becomes a safe reentrant no-op there, and `after_in_parent` restores normal locking
1658
+ # in the parent.
1659
+ os.register_at_fork(
1660
+ before=OutboundSpool._registry_lock.acquire,
1661
+ after_in_parent=OutboundSpool._registry_lock.release,
1662
+ after_in_child=_poison_after_fork,
1663
+ )
1664
+
1665
+
1666
+ # =================================================================================================
1667
+ # Transport -- the HTTP boundary every channel client speaks through, and its production impl.
1668
+ # =================================================================================================
1669
+
1670
+
1671
+ @dataclass(frozen=True)
1672
+ class TransportResponse:
1673
+ status_code: int
1674
+ body: dict[str, Any] | None
1675
+ headers: dict[str, str]
1676
+
1677
+
1678
+ class TransportError(RuntimeError):
1679
+ """Raised by a `Transport.request` implementation when no HTTP response was ever received
1680
+ (connection refused, DNS failure, timeout, ...). `classify_response` treats this identically
1681
+ to an unreachable 5xx -- the guest cannot distinguish "server errored" from "server unreachable"
1682
+ and the contract's retry policy doesn't ask it to."""
1683
+
1684
+
1685
+ class Transport(Protocol):
1686
+ """The seam every channel client is built against, so the fake-platform tests exercise the
1687
+ exact same code path production traffic does -- only what sits behind this protocol differs.
1688
+ """
1689
+
1690
+ def request(
1691
+ self,
1692
+ method: str,
1693
+ url: str,
1694
+ *,
1695
+ headers: dict[str, str],
1696
+ json_body: dict[str, Any] | None = None,
1697
+ data: bytes | Iterator[bytes] | None = None,
1698
+ timeout: float = 30.0,
1699
+ ) -> TransportResponse: ...
1700
+
1701
+
1702
+ class RequestsTransport:
1703
+ """Production `Transport`: a thin `requests.Session` wrapper. Every network-layer failure
1704
+ (`requests.RequestException`, which covers connection errors, timeouts, and retries `requests`
1705
+ itself doesn't handle) is normalized to `TransportError` so `classify_response` never needs to
1706
+ know which HTTP library is underneath."""
1707
+
1708
+ def __init__(self, *, session: requests.Session | None = None) -> None:
1709
+ self._session = session or requests.Session()
1710
+
1711
+ def request(
1712
+ self,
1713
+ method: str,
1714
+ url: str,
1715
+ *,
1716
+ headers: dict[str, str],
1717
+ json_body: dict[str, Any] | None = None,
1718
+ data: bytes | Iterator[bytes] | None = None,
1719
+ timeout: float = 30.0,
1720
+ ) -> TransportResponse:
1721
+ try:
1722
+ response = self._session.request(
1723
+ method, url, headers=headers, json=json_body, data=data, timeout=timeout
1724
+ )
1725
+ except requests.RequestException as exc:
1726
+ raise TransportError(str(exc)) from exc
1727
+ try:
1728
+ body = response.json() if response.content else None
1729
+ except ValueError:
1730
+ body = None
1731
+ return TransportResponse(
1732
+ status_code=response.status_code, body=body, headers=dict(response.headers)
1733
+ )
1734
+
1735
+
1736
+ def _iter_chunks(data: bytes, chunk_size: int) -> Iterator[bytes]:
1737
+ """§3a: "Uploads >64 MB use chunked transfer." A fresh generator is built per send attempt
1738
+ (never reused across a retry) -- a generator is single-use, and reusing an exhausted one would
1739
+ silently upload an empty body on the second attempt."""
1740
+ for start in range(0, len(data), chunk_size):
1741
+ yield data[start : start + chunk_size]
1742
+
1743
+
1744
+ def _parse_retry_after(headers: Mapping[str, str] | None) -> float | None:
1745
+ """ "429 -> honor `Retry-After`." Only the delta-seconds form is parsed (the integer count of
1746
+ seconds to wait) -- the contract never mentions the alternative HTTP-date form and every
1747
+ platform emitter in this ecosystem is expected to send the simple form; an unparseable value is
1748
+ treated as absent so the caller falls back to the computed backoff rather than crashing.
1749
+
1750
+ N5: a negative value is ALSO treated as absent -- `time.sleep(-5)` raises `ValueError`, and a
1751
+ server sending a negative `Retry-After` is malformed input this module owes no obedience to.
1752
+ The upper clamp (`[0, retry_policy.max_backoff_seconds]`) needs the policy, which isn't
1753
+ available here -- `_perform_with_retry` applies that half.
1754
+
1755
+ P6: header-name lookup is case-INsensitive (RFC 9110 §5.1 -- field names are case-insensitive).
1756
+ `RequestsTransport` builds `dict(response.headers)` from `requests`' own `CaseInsensitiveDict`,
1757
+ which drops the case-insensitivity and preserves whatever casing the server actually sent -- a
1758
+ plain `.get("Retry-After")` would miss `RETRY-AFTER`/`Retry-after` and silently fall back to
1759
+ computed backoff instead of honoring the server's wait.
1760
+ """
1761
+ if not headers:
1762
+ return None
1763
+ value = next((v for k, v in headers.items() if k.lower() == "retry-after"), None)
1764
+ if value is None:
1765
+ return None
1766
+ try:
1767
+ parsed = float(value)
1768
+ except ValueError:
1769
+ return None
1770
+ return parsed if parsed >= 0 else None
1771
+
1772
+
1773
+ # =================================================================================================
1774
+ # Error map -- the contract's closed status-code vocabulary ("Error responses" / "Failure semantics
1775
+ # summary"), and the shared retry engine every channel client drives it through.
1776
+ # =================================================================================================
1777
+
1778
+
1779
+ class ChannelOutcome(str, Enum):
1780
+ """Every way one outbound HTTP attempt can resolve, per the contract's failure table. Not a
1781
+ contract vocabulary itself (the wire only ever carries a status code + `{error, message,
1782
+ retryable}`) -- this is this module's own closed classification of that table, the thing
1783
+ `classify_response` computes and every client branches on."""
1784
+
1785
+ DELIVERED = "delivered"
1786
+ RETRYABLE = "retryable"
1787
+ FENCED = "fenced"
1788
+ PERMANENT_ITEM = "permanent_item"
1789
+ CHANNEL_FAILED = "channel_failed"
1790
+ BUDGET_EXCEEDED = "budget_exceeded"
1791
+
1792
+
1793
+ @dataclass(frozen=True)
1794
+ class ChannelError:
1795
+ outcome: ChannelOutcome
1796
+ domain: FailureDomain | None
1797
+ code: str
1798
+ message: str
1799
+ retry_after_seconds: float | None = None
1800
+ # N7: the raw status this was classified from (`None` for a `TransportError`/no-response
1801
+ # outcome) -- carried so a caller can recognize a specific status (413, for the events-batch
1802
+ # halving retry) without `ChannelOutcome`/`code` alone being expressive enough for that.
1803
+ status_code: int | None = None
1804
+
1805
+
1806
+ class HostedFencedError(RuntimeError):
1807
+ """401 (expired) / 403 (fence, scope, mismatch): "stop emitting, exit code 3 ... never an infra
1808
+ retry." Raised by the shared retry engine and never retried -- the entrypoint (outside this
1809
+ module) is the one that translates this into the process exit code."""
1810
+
1811
+ def __init__(self, error: ChannelError) -> None:
1812
+ self.error = error
1813
+ super().__init__(f"{error.code}: {error.message}")
1814
+
1815
+
1816
+ class HostedChannelFailedError(RuntimeError):
1817
+ """404, retried 3x per the contract, still 404: "finalize `platform_sync`." Raised by the
1818
+ shared retry engine once `classify_response` reaches the third 404 attempt."""
1819
+
1820
+ def __init__(self, error: ChannelError) -> None:
1821
+ self.error = error
1822
+ super().__init__(f"{error.code}: {error.message}")
1823
+
1824
+
1825
+ class HostedAttemptSupersededError(RuntimeError):
1826
+ """N22: `409 attempt_superseded` folded into the `ChannelState` latch below -- "a fenced
1827
+ attempt's in-flight requests cannot land after registration of its successor" is a fence in
1828
+ substance, even though the ONE request that received it is still correctly classified
1829
+ `PERMANENT_ITEM` (contract-correct: 409 is item-level, not fence-level). Only raised by
1830
+ `ChannelState.check()` on a LATER call, once a prior call has already seen this code -- the
1831
+ call that actually observed the 409 still returns its normal item-level result."""
1832
+
1833
+ def __init__(self, error: ChannelError) -> None:
1834
+ self.error = error
1835
+ super().__init__(f"{error.code}: {error.message}")
1836
+
1837
+
1838
+ class ChannelState:
1839
+ """N10: shared "stop emitting" latch across the three channel clients for one attempt -- a
1840
+ fence (401/403) or an exhausted channel (404x3) on ANY channel must stop ALL of them, since
1841
+ the token/fence is per-ATTEMPT, not per-channel ("stop emitting ... never an infra retry").
1842
+ Also carries the N22 attempt-supersession latch (409 `attempt_superseded`).
1843
+
1844
+ Construct ONE `ChannelState` per attempt and pass it to `EventsClient`/`ResultsClient`/
1845
+ `ArtifactsClient` alike (each defaults to a private one if not given, which only latches
1846
+ itself -- correct for a single-channel caller, but callers driving more than one channel for
1847
+ the same attempt MUST share one instance to get the cross-channel guarantee this class exists
1848
+ for). Once latched, `check()` raises the SAME error on every subsequent call, from any client
1849
+ sharing this state, without ever touching the transport.
1850
+ """
1851
+
1852
+ def __init__(self) -> None:
1853
+ self._error: (
1854
+ HostedFencedError
1855
+ | HostedChannelFailedError
1856
+ | HostedAttemptSupersededError
1857
+ | None
1858
+ ) = None
1859
+ self._lock = RLock()
1860
+
1861
+ def check(self) -> None:
1862
+ with self._lock:
1863
+ error = self._error
1864
+ if error is not None:
1865
+ raise error
1866
+
1867
+ def latch(
1868
+ self,
1869
+ exc: "HostedFencedError | HostedChannelFailedError | HostedAttemptSupersededError",
1870
+ ) -> None:
1871
+ with self._lock:
1872
+ if self._error is None:
1873
+ self._error = exc
1874
+
1875
+
1876
+ def classify_response(
1877
+ status_code: int | None,
1878
+ body: dict[str, Any] | None,
1879
+ *,
1880
+ attempt: int,
1881
+ retry_after_seconds: float | None = None,
1882
+ ) -> ChannelError | None:
1883
+ """The single call site every channel client classifies a transport outcome through. Returns
1884
+ `None` for a delivered (2xx) response, a `ChannelError` for everything else. `status_code=None`
1885
+ means `TransportError` (no response was ever received) -- classified exactly like an
1886
+ unreachable 5xx. For the 404 branch specifically, `attempt` must be the count of 404 RESPONSES
1887
+ seen so far in this call (not the overall attempt number) -- `_perform_with_retry` tracks that
1888
+ counter separately (N6) so an interleaved 5xx never shortens the 404 budget; every other branch
1889
+ ignores `attempt` entirely.
1890
+
1891
+ N28: the error body's `retryable` field is deliberately never read here -- classification is
1892
+ status-keyed throughout this module (every branch below is keyed on `status_code` alone), which
1893
+ is defensible per the contract's own closed, status-code-driven failure table; if the platform
1894
+ ever marks an unexpected status `retryable` the guest disagrees silently, a known, accepted gap.
1895
+
1896
+ Every branch below is transcribed directly from the contract's "Error responses" paragraph and
1897
+ "Failure semantics summary" table -- see `outbound-channels.md` v1.3 for the prose this mirrors.
1898
+ """
1899
+ if status_code is not None and 200 <= status_code < 300:
1900
+ return None
1901
+
1902
+ error_code = (body or {}).get("error") if body else None
1903
+ message = (body or {}).get("message", "") if body else ""
1904
+
1905
+ if status_code is None:
1906
+ return ChannelError(
1907
+ ChannelOutcome.RETRYABLE,
1908
+ FailureDomain.CONNECTIVITY,
1909
+ error_code or "network_error",
1910
+ message or "transport failure: no response received",
1911
+ status_code=None,
1912
+ )
1913
+ if status_code in (401, 403):
1914
+ return ChannelError(
1915
+ ChannelOutcome.FENCED,
1916
+ None,
1917
+ error_code or "fenced",
1918
+ message,
1919
+ status_code=status_code,
1920
+ )
1921
+ if status_code == 404:
1922
+ # N29: PLATFORM_SYNC on every 404 attempt, not just the third -- a 404 is never a
1923
+ # connectivity fault under §4.6; only the third attempt's outcome is ever surfaced to a
1924
+ # caller, but the domain should not silently disagree across attempts 1-2 vs 3.
1925
+ domain = FailureDomain.PLATFORM_SYNC
1926
+ if attempt < 3:
1927
+ return ChannelError(
1928
+ ChannelOutcome.RETRYABLE,
1929
+ domain,
1930
+ error_code or "not_found",
1931
+ message,
1932
+ status_code=404,
1933
+ )
1934
+ return ChannelError(
1935
+ ChannelOutcome.CHANNEL_FAILED,
1936
+ domain,
1937
+ error_code or "not_found",
1938
+ message,
1939
+ status_code=404,
1940
+ )
1941
+ if status_code == 413:
1942
+ # N7: 413 is Channel 3's artifact-budget code specifically -- only classify it
1943
+ # BUDGET_EXCEEDED when the body actually says so; a 413 on any other channel (e.g. an
1944
+ # events batch that simply exceeded the platform's ingress size limit) falls through to
1945
+ # the unlisted-4xx catch-all below instead of being mislabeled as a budget condition that
1946
+ # channel doesn't have.
1947
+ if error_code == "artifact_budget_exceeded":
1948
+ return ChannelError(
1949
+ ChannelOutcome.BUDGET_EXCEEDED,
1950
+ None,
1951
+ error_code,
1952
+ message,
1953
+ status_code=413,
1954
+ )
1955
+ return ChannelError(
1956
+ ChannelOutcome.PERMANENT_ITEM,
1957
+ None,
1958
+ error_code or "http_413",
1959
+ message,
1960
+ status_code=413,
1961
+ )
1962
+ if status_code == 429:
1963
+ return ChannelError(
1964
+ ChannelOutcome.RETRYABLE,
1965
+ FailureDomain.CONNECTIVITY,
1966
+ error_code or "rate_limited",
1967
+ message,
1968
+ retry_after_seconds=retry_after_seconds,
1969
+ status_code=429,
1970
+ )
1971
+ if status_code in (400, 409, 422):
1972
+ return ChannelError(
1973
+ ChannelOutcome.PERMANENT_ITEM,
1974
+ None,
1975
+ error_code or f"http_{status_code}",
1976
+ message,
1977
+ status_code=status_code,
1978
+ )
1979
+ if 500 <= status_code < 600:
1980
+ return ChannelError(
1981
+ ChannelOutcome.RETRYABLE,
1982
+ FailureDomain.CONNECTIVITY,
1983
+ error_code or "server_error",
1984
+ message,
1985
+ status_code=status_code,
1986
+ )
1987
+ if 400 <= status_code < 500:
1988
+ # "Catch-all: any unlisted 4xx is permanent for that item."
1989
+ return ChannelError(
1990
+ ChannelOutcome.PERMANENT_ITEM,
1991
+ None,
1992
+ error_code or f"http_{status_code}",
1993
+ message,
1994
+ status_code=status_code,
1995
+ )
1996
+ # N27: no other status family is contractual -- notably a 3xx, which should never occur (every
1997
+ # endpoint ends in "/" precisely so Django's POST-redirect problem never arises). Treat as
1998
+ # permanent rather than retrying an endpoint misconfiguration `max_attempts` times before
1999
+ # giving up anyway.
2000
+ return ChannelError(
2001
+ ChannelOutcome.PERMANENT_ITEM,
2002
+ None,
2003
+ error_code or f"http_{status_code}",
2004
+ message,
2005
+ status_code=status_code,
2006
+ )
2007
+
2008
+
2009
+ def compute_backoff_seconds(
2010
+ attempt: int,
2011
+ *,
2012
+ initial_backoff_seconds: float,
2013
+ max_backoff_seconds: float,
2014
+ rng: Callable[[], float] = random.random,
2015
+ ) -> float:
2016
+ """ "retry with backoff (base `retry.initial_backoff_seconds`, cap `retry.max_backoff_seconds`,
2017
+ full jitter)" -- `uniform(0, min(cap, base * 2**(attempt-1)))`. `attempt` is the 1-based attempt
2018
+ that just failed. `rng` is injectable so callers (and tests) can get a deterministic value.
2019
+ """
2020
+ ceiling = min(
2021
+ max_backoff_seconds, initial_backoff_seconds * (2 ** max(0, attempt - 1))
2022
+ )
2023
+ return rng() * ceiling
2024
+
2025
+
2026
+ @dataclass(frozen=True)
2027
+ class RetryPolicy:
2028
+ """Backoff parameters, shared by all three channel clients. Field names mirror
2029
+ `job.HarnessRetryPolicy` deliberately (the contract states these come from the job's own
2030
+ `retry.initial_backoff_seconds`/`retry.max_backoff_seconds`), but this module does not import
2031
+ that class -- a client only needs two floats, and importing the full job-retry model (with its
2032
+ `retryable_domains` field, which governs WHOLE-JOB attempt retries, a distinct concept from a
2033
+ single outbound delivery's backoff) would be a coupling this module doesn't need.
2034
+
2035
+ `max_attempts` bounds one `_perform_with_retry` call's own retry loop for the classes the
2036
+ contract leaves unbounded (network/5xx/429 -- "spool + backoff", no stated attempt ceiling).
2037
+ STUCK DECISION (fail-safe/reversible, contract silent): capped at a generous default (8) rather
2038
+ than looped forever, because durability already lives in the spool/idempotent-wire-design, not
2039
+ in one blocking call -- a caller that wants to keep trying simply invokes the client method
2040
+ again later (`EventsClient.flush()` is designed to be called repeatedly for exactly this
2041
+ reason). Must stay >= 3 for the 404 rule to ever reach its own `CHANNEL_FAILED` transition
2042
+ within a single call; the default comfortably clears that.
2043
+ """
2044
+
2045
+ initial_backoff_seconds: float = 1.0
2046
+ max_backoff_seconds: float = 15.0
2047
+ max_attempts: int = 8
2048
+
2049
+
2050
+ def _perform_with_retry(
2051
+ perform: Callable[[int], TransportResponse],
2052
+ *,
2053
+ retry_policy: RetryPolicy,
2054
+ sleep: Callable[[float], None],
2055
+ rng: Callable[[], float] = random.random,
2056
+ deadline: float | None = None,
2057
+ now: Callable[[], float] = time.monotonic,
2058
+ ) -> tuple[TransportResponse | None, ChannelError | None]:
2059
+ """The one retry/backoff engine all three channel clients drive their single HTTP call through.
2060
+ `perform(attempt)` makes ONE attempt (1-based); this loops it, classifies each outcome via
2061
+ `classify_response`, and:
2062
+
2063
+ - returns `(response, None)` once delivered;
2064
+ - raises `HostedFencedError` immediately on 401/403 (never retried, by contract);
2065
+ - raises `HostedChannelFailedError` once a 404 reaches its third attempt;
2066
+ - returns `(response_or_None, error)` for every other terminal outcome (`PERMANENT_ITEM`,
2067
+ `BUDGET_EXCEEDED`) without retrying -- "deterministic rejections are never retried";
2068
+ - otherwise (`RETRYABLE`) sleeps -- honoring a server `Retry-After` over the computed backoff
2069
+ when present -- and tries again, up to `retry_policy.max_attempts`.
2070
+
2071
+ N5/P5: `deadline` (a `time.monotonic()` value, typically the adapter's flush-window end) bounds
2072
+ ATTEMPT SCHEDULING AND SLEEPS ONLY -- no new attempt starts once `now() >= deadline`, and every
2073
+ sleep (computed backoff OR a server `Retry-After`) is clamped to whatever budget remains. It
2074
+ does NOT bound an attempt's own in-flight request: `perform`'s transport call carries its own,
2075
+ separate `timeout` (the caller's second knob -- `Transport.request`'s `timeout` parameter,
2076
+ unrelated to `deadline`), and nothing here clamps that value against the remaining deadline
2077
+ budget. A caller sizing `deadline` against a hard wall-clock guarantee for the whole call is
2078
+ sizing against a guarantee this function does not provide; only "no new attempt starts, and no
2079
+ sleep runs, once the window is gone" is guaranteed. (P5: widening `perform` to accept a clamped
2080
+ per-attempt timeout was considered and is the more complete fix, but at least one caller outside
2081
+ this module -- `hosted_entrypoint.py`'s `ScenariosClient._post`, which calls this function
2082
+ directly with its own single-argument `perform` closure -- is out of this fix's scope, so
2083
+ changing the call signature here would silently break that caller instead of fixing it. This
2084
+ docstring correction is the floor the round-3 review named for exactly that situation.)
2085
+
2086
+ N6: 404 retries are counted SEPARATELY from the overall attempt number (`not_found_attempts`),
2087
+ so an interleaved 5xx (e.g. `503, 503, 404, 404, 404`) does not shorten the 404 budget -- only
2088
+ three OBSERVED 404s reach `classify_response`'s `CHANNEL_FAILED` transition, regardless of what
2089
+ else happened in between.
2090
+ """
2091
+ attempt = 0
2092
+ not_found_attempts = 0
2093
+ while True:
2094
+ attempt += 1
2095
+ if deadline is not None and now() >= deadline:
2096
+ return None, ChannelError(
2097
+ ChannelOutcome.RETRYABLE,
2098
+ FailureDomain.CONNECTIVITY,
2099
+ "deadline_exceeded",
2100
+ "the flush-window deadline elapsed before this attempt could be made",
2101
+ )
2102
+ try:
2103
+ response = perform(attempt)
2104
+ except TransportError:
2105
+ error = classify_response(None, None, attempt=attempt)
2106
+ response = None
2107
+ else:
2108
+ status = response.status_code
2109
+ if status == 404:
2110
+ not_found_attempts += 1
2111
+ error = classify_response(
2112
+ status,
2113
+ response.body,
2114
+ attempt=not_found_attempts if status == 404 else attempt,
2115
+ retry_after_seconds=_parse_retry_after(response.headers),
2116
+ )
2117
+ if error is None:
2118
+ return response, None
2119
+ if error.outcome is ChannelOutcome.FENCED:
2120
+ raise HostedFencedError(error)
2121
+ if error.outcome is ChannelOutcome.CHANNEL_FAILED:
2122
+ raise HostedChannelFailedError(error)
2123
+ if error.outcome is not ChannelOutcome.RETRYABLE:
2124
+ return response, error
2125
+ if attempt >= retry_policy.max_attempts:
2126
+ return response, error
2127
+ delay = error.retry_after_seconds
2128
+ if delay is not None:
2129
+ delay = min(
2130
+ delay, retry_policy.max_backoff_seconds
2131
+ ) # N5: clamp Retry-After
2132
+ else:
2133
+ delay = compute_backoff_seconds(
2134
+ attempt,
2135
+ initial_backoff_seconds=retry_policy.initial_backoff_seconds,
2136
+ max_backoff_seconds=retry_policy.max_backoff_seconds,
2137
+ rng=rng,
2138
+ )
2139
+ if deadline is not None:
2140
+ remaining = deadline - now()
2141
+ if remaining <= 0:
2142
+ return response, error
2143
+ delay = min(delay, remaining)
2144
+ sleep(delay)
2145
+
2146
+
2147
+ # =================================================================================================
2148
+ # Channel 1 -- Events client. Batches spooled events, advances the watermark only on confirmed
2149
+ # delivery.
2150
+ # =================================================================================================
2151
+
2152
+
2153
+ @dataclass(frozen=True)
2154
+ class EventsFlushResult:
2155
+ delivered_count: int
2156
+ acked_through_sequence: int | None
2157
+ rejected: list[dict[str, Any]]
2158
+ error: ChannelError | None
2159
+ # v1.3 (M7): set when the platform's `acked_through_sequence` fell outside the spool's trusted
2160
+ # range and was ignored rather than trusted -- `error` stays `None` because the HTTP delivery
2161
+ # itself succeeded; only the ack body was untrustworthy.
2162
+ ack_out_of_range: bool = False
2163
+ # N2/N3: set when a 2xx response's `acked_through_sequence` was missing, `null`, or not an
2164
+ # int -- a protocol violation distinct from "present but out of range" above. `error` stays
2165
+ # `None` for the same reason: the HTTP delivery itself succeeded.
2166
+ ack_missing: bool = False
2167
+ # P9: the SpooledRecord bodies dropped this call (a subset of `batch`, keyed by `rejected`),
2168
+ # captured BEFORE `drop_many` removes them from disk. The contract requires a rejected event's
2169
+ # payload be "written to the artifact spool (`log` kind)" -- without this, a caller has no way
2170
+ # to recover that payload at all once `flush()` returns, since the cap applied inside `flush()`
2171
+ # makes it impossible to reliably re-derive which spooled records were even in this batch.
2172
+ dropped_records: list[SpooledRecord] = field(default_factory=list)
2173
+
2174
+
2175
+ _EVENTS_BATCH_PREFIX = (
2176
+ b'{"schema_version":"' + EVENT_SCHEMA_VERSION.encode("utf-8") + b'","events":['
2177
+ )
2178
+ _EVENTS_BATCH_SUFFIX = b"]}"
2179
+
2180
+
2181
+ def _encode_events_batch(records: list[SpooledRecord]) -> bytes:
2182
+ """N19: the contract says "serialize once, spool the bytes, re-send verbatim; never
2183
+ re-serialize on retry" -- for the events BATCH ENVELOPE itself, not just each event inside it.
2184
+ Handing a decoded dict to `json_body=` (the previous shape) let `requests` re-serialize the
2185
+ envelope with its own settings on every send; this instead concatenates the exact bytes
2186
+ `OutboundSpool.append` already wrote for each event, closing the deviation rather than merely
2187
+ documenting it. Safe because `EVENT_SCHEMA_VERSION` is a fixed ASCII constant with no bytes
2188
+ needing escape."""
2189
+ return (
2190
+ _EVENTS_BATCH_PREFIX
2191
+ + b",".join(record.body for record in records)
2192
+ + _EVENTS_BATCH_SUFFIX
2193
+ )
2194
+
2195
+
2196
+ class EventsClient:
2197
+ """Delivers `OutboundSpool`-backed Channel 1 events to `endpoints.events`. One `flush()` call
2198
+ sends one batch (<= `EVENTS_MAX_BATCH` events, <= `EVENTS_MAX_BATCH_BYTES`) of everything
2199
+ spooled since the last confirmed watermark, in spool order
2200
+ (`OutboundSpool.pending_since_watermark`, which reads the log in append/sequence order -- the
2201
+ platform's own ordering rule, "`(attempt_number, sequence)`", so a single-attempt process
2202
+ satisfies it for free). Re-sends spooled bytes verbatim (N19, `_encode_events_batch`) -- never
2203
+ recomputing an event's own `digest`, so "serialize once ... never re-serialize on retry" holds
2204
+ for the one thing that must never drift (the per-event digest, embedded as data).
2205
+
2206
+ `flush()` is meant to be called repeatedly (by whatever background loop owns the call cadence,
2207
+ a scheduler concern outside this module) -- each call is a complete, self-contained delivery
2208
+ attempt (with its own internal retry/backoff via `_perform_with_retry`) that advances the
2209
+ watermark exactly as far as the platform confirmed and leaves everything else spooled for the
2210
+ next call.
2211
+
2212
+ The ack body is UNTRUSTED PLATFORM INPUT end to end (v1.3): `acked_through_sequence` goes
2213
+ through the spool's own M7 clamp (`advance_watermark`); `rejected[]` is filtered to sequences
2214
+ actually present in the batch just sent BEFORE anything is done with it (N1) -- a value the
2215
+ guest never sent cannot cause a drop, and dropping never itself advances the watermark (see
2216
+ `OutboundSpool.drop_many`) -- so the batch-level `advance_watermark(acked_through_sequence)` is
2217
+ the single chokepoint either way.
2218
+ """
2219
+
2220
+ def __init__(
2221
+ self,
2222
+ capabilities: HostedCapabilities,
2223
+ spool: OutboundSpool,
2224
+ transport: Transport | None = None,
2225
+ *,
2226
+ retry_policy: RetryPolicy | None = None,
2227
+ sleep: Callable[[float], None] = time.sleep,
2228
+ rng: Callable[[], float] = random.random,
2229
+ batch_size: int = EVENTS_MAX_BATCH,
2230
+ channel_state: ChannelState | None = None,
2231
+ ) -> None:
2232
+ self._capabilities = capabilities
2233
+ self._spool = spool
2234
+ self._transport = transport or RequestsTransport()
2235
+ self._retry_policy = retry_policy or RetryPolicy()
2236
+ self._sleep = sleep
2237
+ self._rng = rng
2238
+ self._batch_size = max(1, min(batch_size, EVENTS_MAX_BATCH))
2239
+ self._channel_state = channel_state or ChannelState()
2240
+
2241
+ def _cap_batch(self, records: list[SpooledRecord]) -> list[SpooledRecord]:
2242
+ """Proactive half of N7: cap by event count (`_batch_size`) AND cumulative canonical bytes
2243
+ (`EVENTS_MAX_BATCH_BYTES`) before ever building a request -- reduces how often the reactive
2244
+ 413-halving in `flush()` below is ever needed. Always includes at least one record (its own
2245
+ oversized payload is HostedEventDraft's problem, at spool-append time, not this cap's)."""
2246
+ capped = records[: self._batch_size]
2247
+ limited: list[SpooledRecord] = []
2248
+ total_bytes = 0
2249
+ for record in capped:
2250
+ if limited and total_bytes + len(record.body) > EVENTS_MAX_BATCH_BYTES:
2251
+ break
2252
+ limited.append(record)
2253
+ total_bytes += len(record.body)
2254
+ return limited
2255
+
2256
+ def flush(self, *, deadline: float | None = None) -> EventsFlushResult:
2257
+ self._channel_state.check()
2258
+ batch = self._cap_batch(self._spool.pending_since_watermark())
2259
+ if not batch:
2260
+ return EventsFlushResult(
2261
+ delivered_count=0,
2262
+ acked_through_sequence=self._spool.watermark(),
2263
+ rejected=[],
2264
+ error=None,
2265
+ )
2266
+
2267
+ response: TransportResponse | None = None
2268
+ error: ChannelError | None = None
2269
+ while True:
2270
+ body_bytes = _encode_events_batch(batch)
2271
+ headers = {
2272
+ **self._capabilities.auth_headers(),
2273
+ "Content-Type": "application/json",
2274
+ }
2275
+
2276
+ def perform(_attempt: int) -> TransportResponse:
2277
+ return self._transport.request(
2278
+ "POST",
2279
+ self._capabilities.endpoints.events,
2280
+ headers=headers,
2281
+ data=body_bytes,
2282
+ )
2283
+
2284
+ try:
2285
+ response, error = _perform_with_retry(
2286
+ perform,
2287
+ retry_policy=self._retry_policy,
2288
+ sleep=self._sleep,
2289
+ rng=self._rng,
2290
+ deadline=deadline,
2291
+ )
2292
+ except (HostedFencedError, HostedChannelFailedError) as exc:
2293
+ self._channel_state.latch(exc)
2294
+ raise
2295
+ # N7: a 413 on the events channel -- reactively halve the batch and try again rather
2296
+ # than returning a permanent, non-progressing error for the whole thing. Stops once a
2297
+ # single event alone still 413s (defensive; should not happen under EVENT_PAYLOAD_MAX_BYTES).
2298
+ if error is not None and error.status_code == 413 and len(batch) > 1:
2299
+ logger.warning(
2300
+ "events flush: batch of %d events (%d bytes) was rejected with 413 -- halving "
2301
+ "and retrying",
2302
+ len(batch),
2303
+ len(body_bytes),
2304
+ )
2305
+ batch = batch[: len(batch) // 2]
2306
+ continue
2307
+ break
2308
+
2309
+ if error is not None or response is None:
2310
+ return EventsFlushResult(
2311
+ delivered_count=0, acked_through_sequence=None, rejected=[], error=error
2312
+ )
2313
+
2314
+ # N2/N3: the ack body is untrusted platform input, defensively parsed -- a wrong type or a
2315
+ # missing key must never raise out of flush() (that would silence the flusher loop, B1's
2316
+ # outcome by a different route) and must never be treated as "0" (that would silently
2317
+ # re-send the same batch forever, N3).
2318
+ body = response.body if isinstance(response.body, dict) else {}
2319
+ raw_acked = body.get("acked_through_sequence")
2320
+ acked_through: int | None
2321
+ if isinstance(raw_acked, int) and not isinstance(raw_acked, bool):
2322
+ acked_through = raw_acked
2323
+ else:
2324
+ acked_through = None
2325
+ logger.warning(
2326
+ "events flush: 2xx response has a missing/invalid acked_through_sequence (got %r) "
2327
+ "-- treating as a protocol violation, not advancing the watermark",
2328
+ raw_acked,
2329
+ )
2330
+
2331
+ raw_rejected = body.get("rejected")
2332
+ if isinstance(raw_rejected, list):
2333
+ rejected_entries = [
2334
+ entry for entry in raw_rejected if isinstance(entry, dict)
2335
+ ]
2336
+ if len(rejected_entries) != len(raw_rejected):
2337
+ logger.warning(
2338
+ "events flush: rejected[] contained non-object entries -- ignoring them"
2339
+ )
2340
+ else:
2341
+ rejected_entries = []
2342
+ if raw_rejected is not None:
2343
+ logger.warning(
2344
+ "events flush: rejected is %r, not a list -- treating as empty",
2345
+ type(raw_rejected).__name__,
2346
+ )
2347
+
2348
+ if acked_through is None:
2349
+ return EventsFlushResult(
2350
+ delivered_count=0,
2351
+ acked_through_sequence=self._spool.watermark(),
2352
+ rejected=[],
2353
+ error=None,
2354
+ ack_missing=True,
2355
+ )
2356
+
2357
+ # N1: filter rejected[] to sequences the guest ACTUALLY sent in this batch, before doing
2358
+ # anything with them -- a sequence the platform names that was never in `batch` is
2359
+ # untrusted input this module owes no obedience to (it cannot be dropped, since it was
2360
+ # never spooled under that number in the first place, and trusting it would let a
2361
+ # malformed ack orphan pending records by a route the M7 clamp doesn't guard).
2362
+ batch_sequences = {
2363
+ record.sequence for record in batch if record.sequence is not None
2364
+ }
2365
+ valid_rejected: list[dict[str, Any]] = []
2366
+ for entry in rejected_entries:
2367
+ sequence = entry.get("sequence")
2368
+ if (
2369
+ isinstance(sequence, int)
2370
+ and not isinstance(sequence, bool)
2371
+ and sequence in batch_sequences
2372
+ ):
2373
+ valid_rejected.append(entry)
2374
+ else:
2375
+ logger.warning(
2376
+ "events flush: rejected entry names sequence=%r, which was not sent in this "
2377
+ "batch (sent=%s) -- ignoring as untrusted platform input",
2378
+ sequence,
2379
+ sorted(batch_sequences),
2380
+ )
2381
+
2382
+ # P9: capture the dropped records' own bodies BEFORE drop_many physically removes them --
2383
+ # once removed, this is the only place a caller can still recover the payload the contract
2384
+ # requires be "written to the artifact spool (`log` kind)" for a rejected event.
2385
+ rejected_sequences = {entry["sequence"] for entry in valid_rejected}
2386
+ dropped_records = [
2387
+ record for record in batch if record.sequence in rejected_sequences
2388
+ ]
2389
+
2390
+ # N1/N14: pure physical removal, batched into one rewrite -- drop_many never touches the
2391
+ # watermark; the batch-level advance_watermark(acked_through) below is the ONLY chokepoint.
2392
+ self._spool.drop_many(rejected_sequences)
2393
+
2394
+ try:
2395
+ # "the watermark is highest-processed -- accepted AND rejected sequences both advance
2396
+ # it." v1.3: `acked_through_sequence` is untrusted input -- the spool itself enforces
2397
+ # the clamp (M7).
2398
+ self._spool.advance_watermark(acked_through)
2399
+ except OutboundSpoolError:
2400
+ logger.warning(
2401
+ "events flush: platform returned an untrusted acked_through_sequence=%s outside "
2402
+ "the guest's trusted range -- ignoring the ack, watermark unchanged at %s",
2403
+ acked_through,
2404
+ self._spool.watermark(),
2405
+ )
2406
+ return EventsFlushResult(
2407
+ delivered_count=0,
2408
+ acked_through_sequence=self._spool.watermark(),
2409
+ rejected=[],
2410
+ error=None,
2411
+ ack_out_of_range=True,
2412
+ dropped_records=dropped_records,
2413
+ )
2414
+
2415
+ delivered_count = sum(
2416
+ 1
2417
+ for record in batch
2418
+ if record.sequence is not None
2419
+ and record.sequence <= acked_through
2420
+ and record.sequence not in rejected_sequences
2421
+ )
2422
+ return EventsFlushResult(
2423
+ delivered_count=delivered_count,
2424
+ acked_through_sequence=acked_through,
2425
+ rejected=valid_rejected,
2426
+ error=None,
2427
+ dropped_records=dropped_records,
2428
+ )
2429
+
2430
+
2431
+ # =================================================================================================
2432
+ # Channel 2 -- Result receipts. Typed `ResultReceiptDraft` + delivery.
2433
+ # =================================================================================================
2434
+
2435
+
2436
+ class ScenarioStatus(str, Enum):
2437
+ PASSED = "passed"
2438
+ FAILED = "failed"
2439
+ ERRORED = "errored"
2440
+ SKIPPED = "skipped"
2441
+
2442
+
2443
+ class SubGoalResult(BaseModel):
2444
+ model_config = ConfigDict(extra="forbid")
2445
+
2446
+ name: str = Field(min_length=1)
2447
+ held: bool | None
2448
+ reason: str | None
2449
+ judged: bool
2450
+
2451
+
2452
+ class MetricEvaluation(BaseModel):
2453
+ model_config = ConfigDict(extra="forbid")
2454
+
2455
+ name: str = Field(min_length=1)
2456
+ kind: Literal["metric"]
2457
+ score: float = Field(ge=0.0, le=1.0)
2458
+ reason: str
2459
+
2460
+
2461
+ class CheckpointEvaluation(BaseModel):
2462
+ model_config = ConfigDict(extra="forbid")
2463
+
2464
+ name: str = Field(min_length=1)
2465
+ kind: Literal["checkpoint"]
2466
+ passed: bool
2467
+ reason: str
2468
+
2469
+
2470
+ EvaluationResult = MetricEvaluation | CheckpointEvaluation
2471
+
2472
+
2473
+ class CallSummary(BaseModel):
2474
+ model_config = ConfigDict(extra="forbid")
2475
+
2476
+ started_at: str
2477
+ ended_at: str
2478
+ duration_ms: int = Field(ge=0)
2479
+ turns: int = Field(ge=0)
2480
+ transcript_artifact: str | None
2481
+ recording_artifacts: list[str] = Field(default_factory=list)
2482
+ stop_reason: str | None = None
2483
+
2484
+ @model_validator(mode="after")
2485
+ def _validate(self) -> "CallSummary":
2486
+ for label, value in (
2487
+ ("started_at", self.started_at),
2488
+ ("ended_at", self.ended_at),
2489
+ ):
2490
+ if not is_valid_rfc3339_millis(value):
2491
+ raise ValueError(f"call_timestamp_invalid: {label}={value!r}")
2492
+ if self.transcript_artifact is not None and not is_valid_digest(
2493
+ self.transcript_artifact
2494
+ ):
2495
+ raise ValueError(
2496
+ f"call_transcript_artifact_invalid: {self.transcript_artifact!r}"
2497
+ )
2498
+ for artifact in self.recording_artifacts:
2499
+ if not is_valid_digest(artifact):
2500
+ raise ValueError(f"call_recording_artifact_invalid: {artifact!r}")
2501
+ return self
2502
+
2503
+
2504
+ def _unset_default_fields(model: BaseModel, prefix: str = "") -> list[str]:
2505
+ """N23: names the fields a caller did NOT explicitly set (filled from a pydantic default) --
2506
+ the actual shape of a digest-mismatch bug like the review's example: a raw `call` dict that
2507
+ omits `recording_artifacts` computes an external digest over an object without that key, while
2508
+ the model's own re-derivation fills in `recording_artifacts: []`. This is not a general diff
2509
+ against the caller's original raw dict (a model validator has no access to that, only to what
2510
+ pydantic recorded via `model_fields_set`) -- it is the honest, dotted-path subset available from
2511
+ inside the model: which fields with defaults were left unset, one level into nested models
2512
+ (covers `call.recording_artifacts`, not just top-level `world_index`/`schema_version`)."""
2513
+ names: list[str] = []
2514
+ for name in type(model).model_fields:
2515
+ if name == "digest":
2516
+ continue
2517
+ path = f"{prefix}{name}"
2518
+ if name not in model.model_fields_set:
2519
+ names.append(path)
2520
+ value = getattr(model, name)
2521
+ if isinstance(value, BaseModel):
2522
+ names.extend(_unset_default_fields(value, f"{path}."))
2523
+ return names
2524
+
2525
+
2526
+ class ResultReceiptDraft(BaseModel):
2527
+ """Channel 2's wire shape. Mirrors `HostedEventDraft`'s pattern: the caller supplies `digest`
2528
+ (via `build_result_receipt`, computed with `whole_object_digest`), the model re-derives and
2529
+ rejects a mismatch, plus the two exact-shape rules the contract states as literal requirements
2530
+ rather than general validation ("`skipped` receipt body (exact)" and "`errored` receipt body").
2531
+ """
2532
+
2533
+ model_config = ConfigDict(extra="forbid")
2534
+
2535
+ schema_version: str = RESULT_SCHEMA_VERSION
2536
+ job_id: str = Field(min_length=1)
2537
+ attempt_id: str = Field(min_length=1)
2538
+ attempt_number: int = Field(ge=1)
2539
+ scenario_key: str = Field(min_length=1)
2540
+ scenario_id: str = Field(min_length=1)
2541
+ scenario_attempt: Literal[1, 2]
2542
+ world_index: int | None = Field(default=None, ge=0)
2543
+ status: ScenarioStatus
2544
+ sub_goals: list[SubGoalResult]
2545
+ evaluations: list[EvaluationResult]
2546
+ call: CallSummary | None
2547
+ failure: TerminalFailure | None
2548
+ digest: str
2549
+
2550
+ @model_validator(mode="after")
2551
+ def _validate(self) -> "ResultReceiptDraft":
2552
+ if self.schema_version != RESULT_SCHEMA_VERSION:
2553
+ raise ValueError(f"result_schema_unsupported: {self.schema_version}")
2554
+ if not is_valid_digest(self.digest):
2555
+ raise ValueError(f"receipt_digest_invalid: {self.digest!r}")
2556
+ expected_body = self.model_dump(mode="json", exclude={"digest"})
2557
+ # ``stop_reason`` was added after the initial receipt protocol. Preserve
2558
+ # byte-for-byte compatibility for callers that omit it, while including
2559
+ # it in both the digest and wire body whenever it is explicitly supplied.
2560
+ if self.call is not None and "stop_reason" not in self.call.model_fields_set:
2561
+ expected_body["call"].pop("stop_reason", None)
2562
+ expected = whole_object_digest(expected_body)
2563
+ if self.digest != expected:
2564
+ unset = _unset_default_fields(self)
2565
+ hint = (
2566
+ f" -- fields not explicitly set, filled from defaults: {', '.join(unset)}"
2567
+ if unset
2568
+ else ""
2569
+ )
2570
+ raise ValueError(f"receipt_digest_mismatch{hint}")
2571
+
2572
+ if self.status is ScenarioStatus.SKIPPED:
2573
+ if (
2574
+ self.scenario_attempt != 1
2575
+ or self.world_index is not None
2576
+ or self.sub_goals
2577
+ or self.evaluations
2578
+ or self.call is not None
2579
+ or self.failure is not None
2580
+ ):
2581
+ raise ValueError("skipped_receipt_shape_invalid")
2582
+ elif self.status is ScenarioStatus.ERRORED and self.failure is None:
2583
+ raise ValueError("errored_receipt_requires_failure")
2584
+ return self
2585
+
2586
+
2587
+ def build_result_receipt(
2588
+ *,
2589
+ job_id: str,
2590
+ attempt_id: str,
2591
+ attempt_number: int,
2592
+ scenario_key: str,
2593
+ scenario_id: str,
2594
+ scenario_attempt: Literal[1, 2],
2595
+ world_index: int | None,
2596
+ status: ScenarioStatus,
2597
+ sub_goals: list[dict[str, Any]],
2598
+ evaluations: list[dict[str, Any]],
2599
+ call: dict[str, Any] | None,
2600
+ failure: dict[str, Any] | None,
2601
+ extra_secret_values: tuple[str, ...] = (),
2602
+ ) -> dict[str, Any]:
2603
+ """Validate one receipt's shape and compute its digest, returning a plain wire-ready dict.
2604
+ Mirrors `build_event_record`'s contract exactly: callers must pass already wire-typed values
2605
+ (e.g. a metric `score` as `float`, never `int`) -- the digest is computed on the RAW input
2606
+ before model validation/coercion, so a type looseness here fails loudly as a digest mismatch
2607
+ rather than silently spooling a digest that doesn't match what gets sent.
2608
+
2609
+ N9: `redact_outbound_text` runs on `sub_goals[].reason`, `evaluations[].reason`, and
2610
+ `failure.{code,message}` (P8) BEFORE the digest is computed -- same ordering rationale as
2611
+ `build_event_record`: the embedded digest must match the redacted bytes actually sent.
2612
+ """
2613
+
2614
+ def _redact(text: Any) -> Any:
2615
+ return (
2616
+ redact_outbound_text(text, extra_secret_values)
2617
+ if isinstance(text, str)
2618
+ else text
2619
+ )
2620
+
2621
+ sub_goals = [
2622
+ {**goal, "reason": _redact(goal.get("reason"))}
2623
+ if isinstance(goal, dict)
2624
+ else goal
2625
+ for goal in sub_goals
2626
+ ]
2627
+ evaluations = [
2628
+ {**item, "reason": _redact(item.get("reason"))}
2629
+ if isinstance(item, dict)
2630
+ else item
2631
+ for item in evaluations
2632
+ ]
2633
+ if isinstance(failure, dict):
2634
+ updates = {
2635
+ key: _redact(failure[key])
2636
+ for key in ("code", "message")
2637
+ if isinstance(failure.get(key), str)
2638
+ }
2639
+ if updates:
2640
+ failure = {**failure, **updates}
2641
+
2642
+ core: dict[str, Any] = {
2643
+ "schema_version": RESULT_SCHEMA_VERSION,
2644
+ "job_id": job_id,
2645
+ "attempt_id": attempt_id,
2646
+ "attempt_number": attempt_number,
2647
+ "scenario_key": scenario_key,
2648
+ "scenario_id": scenario_id,
2649
+ "scenario_attempt": scenario_attempt,
2650
+ "world_index": world_index,
2651
+ "status": status.value if isinstance(status, ScenarioStatus) else status,
2652
+ "sub_goals": sub_goals,
2653
+ "evaluations": evaluations,
2654
+ "call": call,
2655
+ "failure": failure,
2656
+ }
2657
+ digest = whole_object_digest(core)
2658
+ draft = ResultReceiptDraft.model_validate({**core, "digest": digest})
2659
+ wire = draft.model_dump(mode="json")
2660
+ if call is not None and "stop_reason" not in call:
2661
+ wire["call"].pop("stop_reason", None)
2662
+ return wire
2663
+
2664
+
2665
+ def build_skipped_receipt(
2666
+ *,
2667
+ job_id: str,
2668
+ attempt_id: str,
2669
+ attempt_number: int,
2670
+ scenario_key: str,
2671
+ scenario_id: str,
2672
+ ) -> dict[str, Any]:
2673
+ """The "exact" synthesized body for a scenario that never ran ("The guest synthesizes these
2674
+ during the flush window; the finalizer backfills any still missing")."""
2675
+ return build_result_receipt(
2676
+ job_id=job_id,
2677
+ attempt_id=attempt_id,
2678
+ attempt_number=attempt_number,
2679
+ scenario_key=scenario_key,
2680
+ scenario_id=scenario_id,
2681
+ scenario_attempt=1,
2682
+ world_index=None,
2683
+ status=ScenarioStatus.SKIPPED,
2684
+ sub_goals=[],
2685
+ evaluations=[],
2686
+ call=None,
2687
+ failure=None,
2688
+ )
2689
+
2690
+
2691
+ @dataclass(frozen=True)
2692
+ class ReceiptPushResult:
2693
+ delivered: bool
2694
+ already_existed: bool
2695
+ error: ChannelError | None
2696
+
2697
+
2698
+ class ResultsClient:
2699
+ """Delivers one Channel 2 receipt per call to `endpoints.results`. No spool/watermark of its
2700
+ own -- unlike events, receipts carry no `sequence`; their idempotency key is `(job_id,
2701
+ scenario_key)` (contract), so redelivery safety comes from the wire protocol itself (`200`
2702
+ duplicate on a matching digest) rather than from a local ack cursor. A caller that wants
2703
+ durable at-least-once delivery across a process crash owns that queuing (e.g. an
2704
+ `OutboundSpool(sequenced=False)`, exposed by this module for exactly this) and simply calls
2705
+ `push()` again for anything not yet confirmed -- safe because the platform's own idempotency
2706
+ check is what makes a redelivery a no-op, not any state this client keeps.
2707
+ """
2708
+
2709
+ def __init__(
2710
+ self,
2711
+ capabilities: HostedCapabilities,
2712
+ transport: Transport | None = None,
2713
+ *,
2714
+ retry_policy: RetryPolicy | None = None,
2715
+ sleep: Callable[[float], None] = time.sleep,
2716
+ rng: Callable[[], float] = random.random,
2717
+ channel_state: ChannelState | None = None,
2718
+ ) -> None:
2719
+ self._capabilities = capabilities
2720
+ self._transport = transport or RequestsTransport()
2721
+ self._retry_policy = retry_policy or RetryPolicy()
2722
+ self._sleep = sleep
2723
+ self._rng = rng
2724
+ self._channel_state = channel_state or ChannelState()
2725
+
2726
+ def push(
2727
+ self, receipt: dict[str, Any], *, deadline: float | None = None
2728
+ ) -> ReceiptPushResult:
2729
+ self._channel_state.check()
2730
+
2731
+ def perform(_attempt: int) -> TransportResponse:
2732
+ return self._transport.request(
2733
+ "POST",
2734
+ self._capabilities.endpoints.results,
2735
+ headers=self._capabilities.auth_headers(),
2736
+ json_body=receipt,
2737
+ )
2738
+
2739
+ try:
2740
+ response, error = _perform_with_retry(
2741
+ perform,
2742
+ retry_policy=self._retry_policy,
2743
+ sleep=self._sleep,
2744
+ rng=self._rng,
2745
+ deadline=deadline,
2746
+ )
2747
+ except (HostedFencedError, HostedChannelFailedError) as exc:
2748
+ self._channel_state.latch(exc)
2749
+ raise
2750
+ if error is not None and error.code == "attempt_superseded": # N22
2751
+ self._channel_state.latch(HostedAttemptSupersededError(error))
2752
+ if error is not None or response is None:
2753
+ return ReceiptPushResult(
2754
+ delivered=False, already_existed=False, error=error
2755
+ )
2756
+ # N20: `200` is read as "already exists / duplicate" per the contract's idempotency rule;
2757
+ # the contract never states the success code for a genuinely NEW receipt (this module's own
2758
+ # `FakePlatform` uses `201`, unconfirmed against the real platform -- see the review report).
2759
+ return ReceiptPushResult(
2760
+ delivered=True, already_existed=response.status_code == 200, error=None
2761
+ )
2762
+
2763
+
2764
+ # =================================================================================================
2765
+ # Channel 3 -- Artifacts. Content-addressed upload + `ArtifactManifestDraft` + delivery.
2766
+ # =================================================================================================
2767
+
2768
+
2769
+ class ArtifactKind(str, Enum):
2770
+ RECORDING_COMBINED = "recording_combined"
2771
+ RECORDING_STEREO = "recording_stereo"
2772
+ RECORDING_CUSTOMER = "recording_customer"
2773
+ RECORDING_ASSISTANT = "recording_assistant"
2774
+ TRANSCRIPT = "transcript"
2775
+ TOOL_TRACE = "tool_trace"
2776
+ RESULT = "result"
2777
+ BUILD = "build"
2778
+ TRACE = "trace"
2779
+ LOG = "log"
2780
+ OTHER = "other"
2781
+
2782
+
2783
+ _RESERVED_ARTIFACT_KINDS = frozenset(
2784
+ {
2785
+ ArtifactKind.BUILD,
2786
+ ArtifactKind.TRANSCRIPT,
2787
+ ArtifactKind.TOOL_TRACE,
2788
+ ArtifactKind.RESULT,
2789
+ }
2790
+ )
2791
+
2792
+
2793
+ def is_reserved_artifact_kind(kind: ArtifactKind) -> bool:
2794
+ """ "the budget is partitioned by reservation: `build` + `transcript` + `tool_trace` + `result`
2795
+ are reserved (always admitted); recordings next; `trace`/`log`/`other` last.\""""
2796
+ return kind in _RESERVED_ARTIFACT_KINDS
2797
+
2798
+
2799
+ _RECORDING_ARTIFACT_KINDS = frozenset(
2800
+ {
2801
+ ArtifactKind.RECORDING_COMBINED,
2802
+ ArtifactKind.RECORDING_STEREO,
2803
+ ArtifactKind.RECORDING_CUSTOMER,
2804
+ ArtifactKind.RECORDING_ASSISTANT,
2805
+ }
2806
+ )
2807
+
2808
+
2809
+ def priority_class(kind: ArtifactKind) -> int:
2810
+ """N16: the contract's three-tier budget partition as a total order, lower = admitted first --
2811
+ `0` reserved (`is_reserved_artifact_kind`, always admitted), `1` recordings, `2` `trace`/`log`/
2812
+ `other` (admitted last)."""
2813
+ if is_reserved_artifact_kind(kind):
2814
+ return 0
2815
+ if kind in _RECORDING_ARTIFACT_KINDS:
2816
+ return 1
2817
+ return 2
2818
+
2819
+
2820
+ class ArtifactBudgetTracker:
2821
+ """Client-side mirror of "budget = upload admission ... the guest enforces it first": a
2822
+ per-job cumulative cap across attempts, deduplicated by digest. `would_admit` is a pure check a
2823
+ caller makes before calling `ArtifactsClient.upload` for a non-reserved kind; when it returns
2824
+ `False` the upload is skipped (and named in a `log` event -- the caller's job, not this
2825
+ tracker's). This class does not sequence "refused newest-first" itself -- it has no visibility
2826
+ into candidate ordering across scenarios, which only the scheduler (P9) has; it supplies the
2827
+ admission arithmetic that policy is built on.
2828
+
2829
+ `recording_headroom_bytes` (N16, default 0 -- no behavior change unless a caller opts in):
2830
+ bytes of the remaining budget reserved for recordings not yet seen, subtracted from what a
2831
+ `trace`/`log`/`other` (priority class 2) candidate is allowed to consume. This tracker has no
2832
+ visibility into how many recording bytes are still coming (only the scheduler does), so it
2833
+ cannot give a perfect answer -- reserving a caller-supplied headroom is the honest, testable
2834
+ subset of "recordings next; trace/log/other last" this class alone can enforce.
2835
+ """
2836
+
2837
+ def __init__(
2838
+ self, max_artifact_bytes: int, *, recording_headroom_bytes: int = 0
2839
+ ) -> None:
2840
+ self._max_bytes = max_artifact_bytes
2841
+ self._admitted_bytes = 0
2842
+ self._seen_digests: set[str] = set()
2843
+ self._recording_headroom_bytes = recording_headroom_bytes
2844
+
2845
+ def would_admit(self, kind: ArtifactKind, size: int, *, digest: str) -> bool:
2846
+ if digest in self._seen_digests:
2847
+ return True # already counted; a duplicate upload never grows the budget further
2848
+ tier = priority_class(kind)
2849
+ if tier == 0:
2850
+ return True
2851
+ remaining = self._max_bytes - self._admitted_bytes
2852
+ if tier == 2:
2853
+ remaining -= self._recording_headroom_bytes
2854
+ return size <= remaining
2855
+
2856
+ def record(self, kind: ArtifactKind, size: int, *, digest: str) -> None:
2857
+ del kind # reservation already resolved by would_admit; recorded uniformly here
2858
+ if digest in self._seen_digests:
2859
+ return
2860
+ self._seen_digests.add(digest)
2861
+ self._admitted_bytes += size
2862
+
2863
+ @property
2864
+ def admitted_bytes(self) -> int:
2865
+ return self._admitted_bytes
2866
+
2867
+
2868
+ class ArtifactManifestEntry(BaseModel):
2869
+ model_config = ConfigDict(extra="forbid")
2870
+
2871
+ artifact_id: str
2872
+ kind: ArtifactKind
2873
+ size: int = Field(ge=0)
2874
+ scenario_key: str | None = None
2875
+
2876
+ @model_validator(mode="after")
2877
+ def _validate(self) -> "ArtifactManifestEntry":
2878
+ if not is_valid_digest(self.artifact_id):
2879
+ raise ValueError(f"artifact_id_invalid: {self.artifact_id!r}")
2880
+ return self
2881
+
2882
+
2883
+ class ArtifactManifestDraft(BaseModel):
2884
+ model_config = ConfigDict(extra="forbid")
2885
+
2886
+ schema_version: str = MANIFEST_SCHEMA_VERSION
2887
+ job_id: str = Field(min_length=1)
2888
+ attempt_id: str = Field(min_length=1)
2889
+ attempt_number: int = Field(ge=1)
2890
+ entries: list[ArtifactManifestEntry]
2891
+ complete: bool
2892
+ digest: str
2893
+
2894
+ @model_validator(mode="after")
2895
+ def _validate(self) -> "ArtifactManifestDraft":
2896
+ if self.schema_version != MANIFEST_SCHEMA_VERSION:
2897
+ raise ValueError(f"manifest_schema_unsupported: {self.schema_version}")
2898
+ if not is_valid_digest(self.digest):
2899
+ raise ValueError(f"manifest_digest_invalid: {self.digest!r}")
2900
+ expected = whole_object_digest(self.model_dump(mode="json", exclude={"digest"}))
2901
+ if self.digest != expected:
2902
+ unset = _unset_default_fields(self)
2903
+ hint = (
2904
+ f" -- fields not explicitly set, filled from defaults: {', '.join(unset)}"
2905
+ if unset
2906
+ else ""
2907
+ )
2908
+ raise ValueError(f"manifest_digest_mismatch{hint}")
2909
+ return self
2910
+
2911
+
2912
+ def build_artifact_manifest(
2913
+ *,
2914
+ job_id: str,
2915
+ attempt_id: str,
2916
+ attempt_number: int,
2917
+ entries: list[dict[str, Any]],
2918
+ complete: bool,
2919
+ ) -> dict[str, Any]:
2920
+ """Same pattern as `build_result_receipt`/`build_event_record`: digest computed on the raw
2921
+ input, then re-verified by the model that consumes it."""
2922
+ core: dict[str, Any] = {
2923
+ "schema_version": MANIFEST_SCHEMA_VERSION,
2924
+ "job_id": job_id,
2925
+ "attempt_id": attempt_id,
2926
+ "attempt_number": attempt_number,
2927
+ "entries": entries,
2928
+ "complete": complete,
2929
+ }
2930
+ digest = whole_object_digest(core)
2931
+ draft = ArtifactManifestDraft.model_validate({**core, "digest": digest})
2932
+ return draft.model_dump(mode="json")
2933
+
2934
+
2935
+ @dataclass(frozen=True)
2936
+ class ArtifactUploadResult:
2937
+ delivered: bool
2938
+ already_existed: bool
2939
+ error: ChannelError | None
2940
+
2941
+
2942
+ @dataclass(frozen=True)
2943
+ class ManifestPushResult:
2944
+ delivered: bool
2945
+ already_existed: bool
2946
+ error: ChannelError | None
2947
+
2948
+
2949
+ _DEFAULT_ARTIFACT_CONTENT_TYPES: dict[ArtifactKind, str] = {
2950
+ ArtifactKind.RECORDING_COMBINED: "video/mp4",
2951
+ ArtifactKind.RECORDING_STEREO: "video/mp4",
2952
+ ArtifactKind.RECORDING_CUSTOMER: "video/mp4",
2953
+ ArtifactKind.RECORDING_ASSISTANT: "video/mp4",
2954
+ ArtifactKind.TRANSCRIPT: "application/json",
2955
+ ArtifactKind.RESULT: "application/json",
2956
+ }
2957
+
2958
+
2959
+ def _default_content_type(kind: ArtifactKind) -> str:
2960
+ """N17: §3a pins recordings to mp4 and `transcript` to a JSON array; a platform serving these
2961
+ back to a UI needs an accurate `Content-Type`, not a blanket octet-stream."""
2962
+ return _DEFAULT_ARTIFACT_CONTENT_TYPES.get(kind, "application/octet-stream")
2963
+
2964
+
2965
+ def _artifact_content_type(kind: ArtifactKind, data: bytes) -> str:
2966
+ """Return the wire MIME type, preferring the bytes over the nominal format.
2967
+
2968
+ Hosted voice engines currently materialize RIFF/WAVE recordings even though
2969
+ the v1.4 artifact contract's preferred recording format is MP4. Advertising
2970
+ those bytes as ``video/mp4`` makes browsers reject an otherwise valid audio
2971
+ artifact. Keep the contractual default for opaque/test payloads, but sniff
2972
+ the two recording formats we actually support before uploading.
2973
+ """
2974
+ if kind in {
2975
+ ArtifactKind.RECORDING_COMBINED,
2976
+ ArtifactKind.RECORDING_STEREO,
2977
+ ArtifactKind.RECORDING_CUSTOMER,
2978
+ ArtifactKind.RECORDING_ASSISTANT,
2979
+ }:
2980
+ if len(data) >= 12 and data[:4] == b"RIFF" and data[8:12] == b"WAVE":
2981
+ return "audio/wav"
2982
+ if len(data) >= 12 and data[4:8] == b"ftyp":
2983
+ return "video/mp4"
2984
+ return _default_content_type(kind)
2985
+
2986
+
2987
+ class ArtifactsClient:
2988
+ """Content-addressed upload (§3a) + manifest delivery (§3b) to `endpoints.artifacts`.
2989
+
2990
+ `upload` verifies the given bytes actually hash to the claimed `artifact_id` BEFORE ever
2991
+ calling the transport -- a local, zero-cost check that catches a caller bug (wrong id, wrong
2992
+ bytes) without spending a round trip on it; the platform's own `422 digest_mismatch` remains
2993
+ the authority for anything this local check cannot see (partial reads, transport corruption).
2994
+ On a `422 digest_mismatch` FROM THE PLATFORM specifically, the contract grants exactly one
2995
+ extra whole-upload retry ("re-upload once, then the referencing scenario is `errored`") --
2996
+ distinct from `_perform_with_retry`'s own loop, which treats every 422 as `PERMANENT_ITEM` and
2997
+ never retries it; this method wraps that loop in one more, narrower retry layer that fires only
2998
+ for that one code.
2999
+
3000
+ Size accounting: `X-Artifact-Size` is never a caller-supplied value -- it is always derived
3001
+ from `len(data)`, the same bytes actually transmitted, so a `422 size_mismatch` against what
3002
+ this client sends is structurally unreachable from here (the platform's own count remains the
3003
+ authority for what actually arrived over the wire).
3004
+
3005
+ N18: once a 413 `artifact_budget_exceeded` is observed, this instance latches locally -- every
3006
+ later `upload()` for a NON-reserved kind is refused without contacting the platform at all
3007
+ ("stop uploading non-reserved kinds, log, continue the run"); reserved kinds keep uploading
3008
+ (they are always admitted, budget or not).
3009
+ """
3010
+
3011
+ def __init__(
3012
+ self,
3013
+ capabilities: HostedCapabilities,
3014
+ transport: Transport | None = None,
3015
+ *,
3016
+ retry_policy: RetryPolicy | None = None,
3017
+ sleep: Callable[[float], None] = time.sleep,
3018
+ rng: Callable[[], float] = random.random,
3019
+ chunk_threshold_bytes: int = ARTIFACT_CHUNKED_UPLOAD_THRESHOLD_BYTES,
3020
+ chunk_size_bytes: int = 8 * 1024 * 1024,
3021
+ channel_state: ChannelState | None = None,
3022
+ ) -> None:
3023
+ self._capabilities = capabilities
3024
+ self._transport = transport or RequestsTransport()
3025
+ self._retry_policy = retry_policy or RetryPolicy()
3026
+ self._sleep = sleep
3027
+ self._rng = rng
3028
+ self._chunk_threshold_bytes = chunk_threshold_bytes
3029
+ self._chunk_size_bytes = chunk_size_bytes
3030
+ self._channel_state = channel_state or ChannelState()
3031
+ self._budget_exhausted = False # N18
3032
+
3033
+ def upload(
3034
+ self,
3035
+ artifact_id_hex: str,
3036
+ data: bytes,
3037
+ *,
3038
+ kind: ArtifactKind,
3039
+ scenario_key: str | None = None,
3040
+ content_type: str | None = None,
3041
+ deadline: float | None = None,
3042
+ ) -> ArtifactUploadResult:
3043
+ self._channel_state.check()
3044
+ if self._budget_exhausted and not is_reserved_artifact_kind(kind):
3045
+ logger.warning(
3046
+ "artifacts upload: budget already exhausted (413 artifact_budget_exceeded observed "
3047
+ "earlier this attempt) -- skipping non-reserved kind=%s without contacting the "
3048
+ "platform",
3049
+ kind.value,
3050
+ )
3051
+ return ArtifactUploadResult(
3052
+ delivered=False,
3053
+ already_existed=False,
3054
+ error=ChannelError(
3055
+ ChannelOutcome.BUDGET_EXCEEDED,
3056
+ None,
3057
+ "artifact_budget_exceeded",
3058
+ "budget already exhausted for this attempt (latched locally)",
3059
+ status_code=None,
3060
+ ),
3061
+ )
3062
+ if not re.fullmatch(r"[0-9a-f]{64}", artifact_id_hex):
3063
+ raise ValueError(f"artifact_id_invalid: {artifact_id_hex!r}")
3064
+ if artifact_id_hex == "manifest":
3065
+ # Unreachable through the hex check above (`manifest` isn't 64 hex chars) but the
3066
+ # contract calls this collision out by name ("`manifest` is a reserved segment").
3067
+ raise ValueError("artifact_id_reserved: manifest")
3068
+ computed = hashlib.sha256(data).hexdigest()
3069
+ if computed != artifact_id_hex:
3070
+ raise ValueError(
3071
+ f"artifact_digest_mismatch_local: expected {artifact_id_hex}, computed {computed}"
3072
+ )
3073
+
3074
+ url = f"{self._capabilities.endpoints.artifacts}{artifact_id_hex}/"
3075
+ size = len(data)
3076
+ headers = {
3077
+ **self._capabilities.auth_headers(),
3078
+ "X-Artifact-Kind": kind.value,
3079
+ "X-Artifact-Size": str(size),
3080
+ "Content-Type": content_type or _artifact_content_type(kind, data),
3081
+ }
3082
+ if scenario_key is not None:
3083
+ headers["X-Scenario-Key"] = scenario_key
3084
+
3085
+ def perform(_attempt: int) -> TransportResponse:
3086
+ # A fresh generator per attempt when chunked -- see `_iter_chunks`.
3087
+ body: bytes | Iterator[bytes] = (
3088
+ _iter_chunks(data, self._chunk_size_bytes)
3089
+ if size > self._chunk_threshold_bytes
3090
+ else data
3091
+ )
3092
+ return self._transport.request("PUT", url, headers=headers, data=body)
3093
+
3094
+ response: TransportResponse | None = None
3095
+ error: ChannelError | None = None
3096
+ for outer_attempt in range(
3097
+ 2
3098
+ ): # "re-upload once" on a platform-confirmed digest mismatch
3099
+ try:
3100
+ response, error = _perform_with_retry(
3101
+ perform,
3102
+ retry_policy=self._retry_policy,
3103
+ sleep=self._sleep,
3104
+ rng=self._rng,
3105
+ deadline=deadline,
3106
+ )
3107
+ except (HostedFencedError, HostedChannelFailedError) as exc:
3108
+ self._channel_state.latch(exc)
3109
+ raise
3110
+ if not (
3111
+ error is not None
3112
+ and error.code == "digest_mismatch"
3113
+ and outer_attempt == 0
3114
+ ):
3115
+ break
3116
+
3117
+ if error is not None and error.outcome is ChannelOutcome.BUDGET_EXCEEDED:
3118
+ self._budget_exhausted = True # N18
3119
+
3120
+ if error is not None or response is None:
3121
+ return ArtifactUploadResult(
3122
+ delivered=False, already_existed=False, error=error
3123
+ )
3124
+ # N20: `200` is read as "already exists" per the contract's content-addressed upload
3125
+ # semantics; the success code for a genuinely NEW upload is `201` (this module's own
3126
+ # `FakePlatform` matches that but it is unconfirmed against the real platform).
3127
+ return ArtifactUploadResult(
3128
+ delivered=True, already_existed=response.status_code == 200, error=None
3129
+ )
3130
+
3131
+ def push_manifest(
3132
+ self, manifest: dict[str, Any], *, deadline: float | None = None
3133
+ ) -> ManifestPushResult:
3134
+ self._channel_state.check()
3135
+ url = f"{self._capabilities.endpoints.artifacts}manifest/"
3136
+
3137
+ def perform(_attempt: int) -> TransportResponse:
3138
+ return self._transport.request(
3139
+ "POST",
3140
+ url,
3141
+ headers=self._capabilities.auth_headers(),
3142
+ json_body=manifest,
3143
+ )
3144
+
3145
+ try:
3146
+ response, error = _perform_with_retry(
3147
+ perform,
3148
+ retry_policy=self._retry_policy,
3149
+ sleep=self._sleep,
3150
+ rng=self._rng,
3151
+ deadline=deadline,
3152
+ )
3153
+ except (HostedFencedError, HostedChannelFailedError) as exc:
3154
+ self._channel_state.latch(exc)
3155
+ raise
3156
+ if error is not None and error.code == "attempt_superseded": # N22
3157
+ self._channel_state.latch(HostedAttemptSupersededError(error))
3158
+ if error is not None or response is None:
3159
+ return ManifestPushResult(
3160
+ delivered=False, already_existed=False, error=error
3161
+ )
3162
+ # N20: same caveat as receipts/uploads above -- `200` == duplicate is contract-stated,
3163
+ # the new-manifest success code is `201` per `FakePlatform`, unconfirmed against the real
3164
+ # platform.
3165
+ return ManifestPushResult(
3166
+ delivered=True, already_existed=response.status_code == 200, error=None
3167
+ )
3168
+
3169
+
3170
+ __all__ = [
3171
+ "ARTIFACT_CHUNKED_UPLOAD_THRESHOLD_BYTES",
3172
+ "CAPABILITIES_PATH",
3173
+ "CAPABILITIES_SCHEMA_VERSION",
3174
+ "EVENTS_MAX_BATCH",
3175
+ "EVENTS_MAX_BATCH_BYTES",
3176
+ "EVENT_PAYLOAD_MAX_BYTES",
3177
+ "EVENT_SCHEMA_VERSION",
3178
+ "FLUSH_WINDOW_SECONDS",
3179
+ "MANIFEST_SCHEMA_VERSION",
3180
+ "RESULT_SCHEMA_VERSION",
3181
+ "ArtifactBudgetTracker",
3182
+ "ArtifactKind",
3183
+ "ArtifactManifestDraft",
3184
+ "ArtifactManifestEntry",
3185
+ "ArtifactUploadResult",
3186
+ "ArtifactsClient",
3187
+ "BaselineFrozenPayload",
3188
+ "BaselineInputsChangedPayload",
3189
+ "CallSummary",
3190
+ "CapabilitiesError",
3191
+ "ChannelError",
3192
+ "ChannelOutcome",
3193
+ "ChannelState",
3194
+ "CheckpointEvaluation",
3195
+ "DegradeReason",
3196
+ "EvaluationResult",
3197
+ "EventsClient",
3198
+ "EventsFlushResult",
3199
+ "HostedAttemptSupersededError",
3200
+ "HostedCapabilities",
3201
+ "HostedChannelFailedError",
3202
+ "HostedEndpoints",
3203
+ "HostedEvent",
3204
+ "HostedEventDraft",
3205
+ "HostedFencedError",
3206
+ "LogLevel",
3207
+ "LogPayload",
3208
+ "ManifestPushResult",
3209
+ "MetricEvaluation",
3210
+ "OutboundError",
3211
+ "OutboundEventType",
3212
+ "OutboundSpool",
3213
+ "OutboundSpoolError",
3214
+ "ParallelismDegradedPayload",
3215
+ "ReceiptPushResult",
3216
+ "RequestsTransport",
3217
+ "ResultReceiptDraft",
3218
+ "ResultsClient",
3219
+ "RetryPolicy",
3220
+ "ScenarioCounts",
3221
+ "ScenarioRetriedPayload",
3222
+ "ScenarioStartedPayload",
3223
+ "ScenarioStatus",
3224
+ "SpooledRecord",
3225
+ "StageChangedPayload",
3226
+ "SubGoalResult",
3227
+ "TerminalFailure",
3228
+ "TerminalPayload",
3229
+ "TerminalReason",
3230
+ "Transport",
3231
+ "TransportError",
3232
+ "TransportResponse",
3233
+ "WorldUnhealthyPayload",
3234
+ "build_artifact_manifest",
3235
+ "build_event_record",
3236
+ "build_result_receipt",
3237
+ "build_skipped_receipt",
3238
+ "canonical_bytes",
3239
+ "classify_response",
3240
+ "compute_backoff_seconds",
3241
+ "event_payload_digest",
3242
+ "format_rfc3339_millis",
3243
+ "is_reserved_artifact_kind",
3244
+ "is_valid_digest",
3245
+ "is_valid_rfc3339_millis",
3246
+ "load_capabilities",
3247
+ "priority_class",
3248
+ "redact_outbound_text",
3249
+ "sha256_digest",
3250
+ "truncate_log_message",
3251
+ "whole_object_digest",
3252
+ ]