agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,3252 @@
|
|
|
1
|
+
"""Outbound reporting — `outbound-channels.md` v1.3, paired with `hosted-execution-seams.md` v1.11
|
|
2
|
+
(the spine). All guest -> platform traffic (events, result receipts, artifacts) is outbound HTTPS;
|
|
3
|
+
the platform never calls in. Two halves:
|
|
4
|
+
|
|
5
|
+
Foundations (part 1): the platform capability declaration the gateway uploads to
|
|
6
|
+
`/run/futureagi/capabilities.json`, the byte-exact canonical serialization every digest in the
|
|
7
|
+
contract is built from, and the durable local spool (with its monotonic sequence allocator) that
|
|
8
|
+
emission sits behind so a killed process within a live sandbox never loses or duplicates a
|
|
9
|
+
record. (A killed sandbox is deleted by the gateway, spool and all -- there is no in-sandbox
|
|
10
|
+
restart producer in the spine today, so the cross-restart recovery machinery guards a scenario
|
|
11
|
+
that isn't triggerable yet, but the in-process failed-write and watermark-durability guarantees
|
|
12
|
+
are load-bearing from the first event.)
|
|
13
|
+
|
|
14
|
+
Transport (part 2): a channel-neutral `Transport` protocol plus a `requests`-backed production
|
|
15
|
+
implementation; the closed status-code error map (`classify_response`) shared by all three channel
|
|
16
|
+
clients; and the clients themselves — `EventsClient` (batches spooled events, advances the spool
|
|
17
|
+
watermark only on confirmed delivery), `ResultsClient` (typed `ResultReceiptDraft` + delivery), and
|
|
18
|
+
`ArtifactsClient` (content-addressed upload + `ArtifactManifestDraft` + delivery), each sharing one
|
|
19
|
+
retry/backoff engine (`_perform_with_retry`) that raises on the two channel-ending outcomes
|
|
20
|
+
(`HostedFencedError` for 401/403, `HostedChannelFailedError` for 404 exhausted) and returns a typed
|
|
21
|
+
`ChannelError` for everything else a caller must log-and-continue on.
|
|
22
|
+
|
|
23
|
+
`HostedEvent`/`HostedEventDraft` model Channel 1's wire shape (the "hosted event model" the spine's
|
|
24
|
+
implementation-delta list calls for — `sequence`/`attempt_id`/`attempt_number`/`stage`/`digest` —
|
|
25
|
+
distinct from `fi.simulate.runtime.events.CanonicalEvent`, which remains the local-SDK wire and is
|
|
26
|
+
untouched by this module).
|
|
27
|
+
|
|
28
|
+
Redaction (v1.3 Channel 1; seams v1.11 §3): `redact_outbound_text` scrubs URL userinfo
|
|
29
|
+
(`scheme://user:pw@` -> `scheme://user:***@`) plus an adapter-supplied secret-value list, applied
|
|
30
|
+
inside `build_event_record`/`build_result_receipt` to every free-text field the contract names
|
|
31
|
+
(`log.message`, `world_unhealthy.cause`, `terminal`/receipt `failure.message`, sub_goal/evaluation
|
|
32
|
+
`reason`). It is NOT the full "same secret-content scan as the artifact sealer" the contract also
|
|
33
|
+
requires — that broader scan is a separate, sealer-side obligation this module does not implement.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import hashlib
|
|
39
|
+
import json
|
|
40
|
+
import logging
|
|
41
|
+
import os
|
|
42
|
+
import random
|
|
43
|
+
import re
|
|
44
|
+
import time
|
|
45
|
+
import uuid
|
|
46
|
+
from collections.abc import Callable, Collection, Iterator, Mapping
|
|
47
|
+
from dataclasses import dataclass, field
|
|
48
|
+
from datetime import datetime, timezone
|
|
49
|
+
from enum import Enum
|
|
50
|
+
from pathlib import Path
|
|
51
|
+
from threading import RLock
|
|
52
|
+
from typing import Annotated, Any, ClassVar, Literal, Protocol
|
|
53
|
+
from urllib.parse import urlparse
|
|
54
|
+
|
|
55
|
+
try: # N30: fcntl is POSIX-only; the guest is Linux-only, but the module must still IMPORT
|
|
56
|
+
import fcntl
|
|
57
|
+
except ImportError: # pragma: no cover - non-POSIX
|
|
58
|
+
fcntl = None # type: ignore[assignment]
|
|
59
|
+
|
|
60
|
+
import requests
|
|
61
|
+
from pydantic import (
|
|
62
|
+
AfterValidator,
|
|
63
|
+
BaseModel,
|
|
64
|
+
ConfigDict,
|
|
65
|
+
Field,
|
|
66
|
+
JsonValue,
|
|
67
|
+
ValidationError,
|
|
68
|
+
field_serializer,
|
|
69
|
+
model_validator,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
from .job import FailureDomain, HarnessStage
|
|
73
|
+
|
|
74
|
+
logger = logging.getLogger(__name__)
|
|
75
|
+
|
|
76
|
+
CAPABILITIES_SCHEMA_VERSION = "futureagi.harness-capabilities.v1"
|
|
77
|
+
CAPABILITIES_PATH = "/run/futureagi/capabilities.json"
|
|
78
|
+
EVENT_SCHEMA_VERSION = "futureagi.harness-event.v1"
|
|
79
|
+
RESULT_SCHEMA_VERSION = "futureagi.harness-result.v1"
|
|
80
|
+
MANIFEST_SCHEMA_VERSION = "futureagi.harness-manifest.v1"
|
|
81
|
+
|
|
82
|
+
# "Channel 1" limits (outbound-channels.md v1.3) that every producer of an event record needs
|
|
83
|
+
# to honor before it ever reaches a transport client. MIN-12: consumed by EventsClient.flush()
|
|
84
|
+
# (stamps `schema_version`, clamps the batch to EVENTS_MAX_BATCH) and by HostedEventDraft's own
|
|
85
|
+
# size check -- no longer just declared and unused.
|
|
86
|
+
EVENTS_MAX_BATCH = 100
|
|
87
|
+
EVENT_PAYLOAD_MAX_BYTES = 32 * 1024
|
|
88
|
+
# N7: a cumulative-bytes cap on top of EVENTS_MAX_BATCH's event-count cap. 100 events * 32KB could
|
|
89
|
+
# reach ~3.2MB; this keeps a proactively-built batch comfortably under a common ~1MB ingress cap
|
|
90
|
+
# (nginx's default) so 413 is the exception, not the steady state -- EventsClient.flush() still
|
|
91
|
+
# halves and retries reactively on an observed 413 regardless of this cap.
|
|
92
|
+
EVENTS_MAX_BATCH_BYTES = 900_000
|
|
93
|
+
# §3a: uploads over this size use chunked transfer; consumed by ArtifactsClient's default
|
|
94
|
+
# chunk_threshold_bytes.
|
|
95
|
+
ARTIFACT_CHUNKED_UPLOAD_THRESHOLD_BYTES = 64 * 1024 * 1024
|
|
96
|
+
# "Sequencing"/"Flush window": the drain deadline from the cancel signal, TTL, or terminal event.
|
|
97
|
+
# N5/N24: this module does not compute a deadline from it -- every public client method
|
|
98
|
+
# (EventsClient.flush / ResultsClient.push / ArtifactsClient.upload / .push_manifest) instead
|
|
99
|
+
# accepts an explicit `deadline: float | None` (a `time.monotonic()` value). The adapter (P10) is
|
|
100
|
+
# the one process-wide owner of "when did the window start," so it is the one that turns this
|
|
101
|
+
# constant into the deadline value it passes in -- not this module.
|
|
102
|
+
FLUSH_WINDOW_SECONDS = 120
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# =================================================================================================
|
|
106
|
+
# Canonicalization -- the byte-exact serialization every digest in the contract is built from.
|
|
107
|
+
# =================================================================================================
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class OutboundError(RuntimeError):
|
|
111
|
+
"""Generic typed failure for this module's canonicalization/digest layer -- same `code`/
|
|
112
|
+
`message` shape as `CapabilitiesError`/`OutboundSpoolError`, used where neither of those is the
|
|
113
|
+
right domain (canonicalization itself, and shape checks that run before any spool or
|
|
114
|
+
capabilities object exists)."""
|
|
115
|
+
|
|
116
|
+
def __init__(self, code: str, message: str) -> None:
|
|
117
|
+
self.code = code
|
|
118
|
+
self.message = message
|
|
119
|
+
super().__init__(f"{code}: {message}")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def canonical_bytes(value: Any) -> bytes:
|
|
123
|
+
"""The contract's canonical form ("Canonicalization (every digest in this file)"):
|
|
124
|
+
``json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False)``,
|
|
125
|
+
encoded UTF-8 (v1.3 pins `allow_nan=False` explicitly). This is the ONLY place that call is
|
|
126
|
+
made -- every digest function in this module goes through it, so a change to the algorithm
|
|
127
|
+
cannot happen in only one of them.
|
|
128
|
+
|
|
129
|
+
`allow_nan=False` changes nothing about the bytes for any value that was already valid JSON --
|
|
130
|
+
NaN/Infinity are not RFC 8259, so a float that would have silently produced unparseable bytes
|
|
131
|
+
now fails loudly here instead of downstream at the platform's parser.
|
|
132
|
+
|
|
133
|
+
Never re-derive an already-spooled record's bytes by calling this again on retry: a dict's key
|
|
134
|
+
order is stable within one process but nothing guarantees float formatting or dict construction
|
|
135
|
+
order is bit-identical across a restart. `OutboundSpool` hands back the literal bytes it wrote;
|
|
136
|
+
those are what a retry re-sends, per the contract's "serialize once, spool the bytes, re-send
|
|
137
|
+
verbatim; never re-serialize on retry."
|
|
138
|
+
"""
|
|
139
|
+
try:
|
|
140
|
+
return json.dumps(
|
|
141
|
+
value,
|
|
142
|
+
sort_keys=True,
|
|
143
|
+
separators=(",", ":"),
|
|
144
|
+
ensure_ascii=False,
|
|
145
|
+
allow_nan=False,
|
|
146
|
+
).encode("utf-8")
|
|
147
|
+
except ValueError as exc:
|
|
148
|
+
raise OutboundError("canonical_value_not_finite", str(exc)) from exc
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def sha256_digest(data: bytes) -> str:
|
|
152
|
+
return "sha256:" + hashlib.sha256(data).hexdigest()
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def event_payload_digest(payload: dict[str, Any]) -> str:
|
|
156
|
+
"""Event digest scope: the `payload` object alone (not the envelope around it)."""
|
|
157
|
+
return sha256_digest(canonical_bytes(payload))
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _json_native_offense(value: Any, path: str) -> str | None:
|
|
161
|
+
"""Walks `value` looking for the first thing `canonical_bytes` cannot represent for a reason
|
|
162
|
+
OTHER than NaN/Infinity (which `canonical_bytes` itself catches): a non-JSON-native Python
|
|
163
|
+
value (`datetime`, `Decimal`, `UUID`, ...) or a non-string dict key. Returns the offending key
|
|
164
|
+
path (e.g. `"call.started_at"`), or `None` if the tree is clean."""
|
|
165
|
+
if value is None or isinstance(value, (bool, int, float, str)):
|
|
166
|
+
return None
|
|
167
|
+
if isinstance(value, dict):
|
|
168
|
+
for key, item in value.items():
|
|
169
|
+
if not isinstance(key, str):
|
|
170
|
+
return (
|
|
171
|
+
f"{path}[<non-string-key>:{key!r}]"
|
|
172
|
+
if path
|
|
173
|
+
else f"<non-string-key>:{key!r}"
|
|
174
|
+
)
|
|
175
|
+
offense = _json_native_offense(item, f"{path}.{key}" if path else key)
|
|
176
|
+
if offense is not None:
|
|
177
|
+
return offense
|
|
178
|
+
return None
|
|
179
|
+
if isinstance(value, list):
|
|
180
|
+
for index, item in enumerate(value):
|
|
181
|
+
offense = _json_native_offense(item, f"{path}[{index}]")
|
|
182
|
+
if offense is not None:
|
|
183
|
+
return offense
|
|
184
|
+
return None
|
|
185
|
+
return path or "<root>"
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def whole_object_digest(obj: dict[str, Any]) -> str:
|
|
189
|
+
"""Receipt/manifest digest scope: the whole object with the `digest` key ABSENT.
|
|
190
|
+
|
|
191
|
+
The key is popped, never set to `None` -- the contract is explicit that "absent and null are
|
|
192
|
+
different bytes," so silently keeping `digest: null` in the canonicalized form would compute a
|
|
193
|
+
different (wrong) hash than what the platform verifies against.
|
|
194
|
+
|
|
195
|
+
Unlike an event payload (already pydantic-validated as `dict[str, JsonValue]` before it ever
|
|
196
|
+
reaches `event_payload_digest`), receipts and manifests reach this function as hand-built
|
|
197
|
+
dicts from whatever calls it -- a bare `TypeError` from `json.dumps` on a non-JSON-native value
|
|
198
|
+
or a non-string key is a debugging dead end with no indication of WHERE in the object the bad
|
|
199
|
+
value lives. This walks the tree first and raises a typed `OutboundError` naming the offending
|
|
200
|
+
key path instead.
|
|
201
|
+
"""
|
|
202
|
+
core = {key: value for key, value in obj.items() if key != "digest"}
|
|
203
|
+
offense = _json_native_offense(core, "")
|
|
204
|
+
if offense is not None:
|
|
205
|
+
raise OutboundError(
|
|
206
|
+
"digest_value_not_json_native", f"non-JSON-native value at: {offense}"
|
|
207
|
+
)
|
|
208
|
+
return sha256_digest(canonical_bytes(core))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
_DIGEST_PATTERN = re.compile(r"sha256:[0-9a-f]{64}")
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def is_valid_digest(value: str) -> bool:
|
|
215
|
+
return bool(_DIGEST_PATTERN.fullmatch(value))
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
_RFC3339_MILLIS_PATTERN = re.compile(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z")
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def format_rfc3339_millis(value: datetime) -> str:
|
|
222
|
+
"""The contract's exact timestamp wire form ("Timestamps: RFC 3339, UTC, `Z`, millisecond
|
|
223
|
+
precision"). Unlike `HostedEvent.emitted_at` (an envelope field the event digest scope never
|
|
224
|
+
covers, so its exact string form doesn't matter), receipt/manifest timestamps such as
|
|
225
|
+
`call.started_at` sit INSIDE the `whole_object_digest` scope -- what we hash must be byte-
|
|
226
|
+
identical to what we send, so those fields are plain `str` on the wire models, produced only
|
|
227
|
+
through this function, never through a datetime's default serialization (which pydantic would
|
|
228
|
+
render as e.g. `+00:00` offset and six-digit microseconds, not `Z` and milliseconds).
|
|
229
|
+
"""
|
|
230
|
+
if value.tzinfo is None:
|
|
231
|
+
raise ValueError("naive_datetime_not_allowed")
|
|
232
|
+
utc = value.astimezone(timezone.utc)
|
|
233
|
+
return utc.strftime("%Y-%m-%dT%H:%M:%S.") + f"{utc.microsecond // 1000:03d}Z"
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def is_valid_rfc3339_millis(value: str) -> bool:
|
|
237
|
+
return bool(_RFC3339_MILLIS_PATTERN.fullmatch(value))
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _require_utc_millis(value: datetime) -> datetime:
|
|
241
|
+
"""Shared `AfterValidator` for every `datetime` field this module serializes through
|
|
242
|
+
`format_rfc3339_millis` (`HostedEventDraft.emitted_at`, `HostedCapabilities.expires_at`):
|
|
243
|
+
rejects a naive datetime outright (the contract's wire form has no naive representation),
|
|
244
|
+
converts any other offset to UTC, and truncates to millisecond precision so the VALUE itself --
|
|
245
|
+
not just its string rendering -- matches what gets sent on the wire."""
|
|
246
|
+
if value.tzinfo is None:
|
|
247
|
+
raise ValueError("naive_datetime_not_allowed")
|
|
248
|
+
utc = value.astimezone(timezone.utc)
|
|
249
|
+
return utc.replace(microsecond=(utc.microsecond // 1000) * 1000)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
UtcMillisDatetime = Annotated[datetime, AfterValidator(_require_utc_millis)]
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
# N9/P3: URL userinfo -- `scheme://user:pw@host` -> `scheme://user:***@host`, matching the seams
|
|
256
|
+
# contract's own example (`postgresql://harness:***@...`). The username group is now OPTIONAL
|
|
257
|
+
# (`redis://:pw@host`, the canonical empty-username shape for Redis/RabbitMQ/Mongo, is a real
|
|
258
|
+
# managed-store DSN form) and so is the whole password group (`https://<token>@host`, the standard
|
|
259
|
+
# way a bearer token appears in git/registry output) -- a bare userinfo token is masked outright
|
|
260
|
+
# rather than left verbatim on the theory that "a username alone is not a secret," which is false
|
|
261
|
+
# for a token.
|
|
262
|
+
_USERINFO_PATTERN = re.compile(
|
|
263
|
+
r"([a-zA-Z][a-zA-Z0-9+.\-]*://)([^\s:/?#@]*)(:[^\s/?#]*)?@"
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _mask_userinfo(match: re.Match[str]) -> str:
|
|
268
|
+
scheme, user, password = match.group(1), match.group(2), match.group(3)
|
|
269
|
+
return f"{scheme}{user}:***@" if password is not None else f"{scheme}***@"
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def redact_outbound_text(value: str, extra_secret_values: tuple[str, ...] = ()) -> str:
|
|
273
|
+
"""Scrubs a single free-text field before it can leave the sandbox on any of the three
|
|
274
|
+
channels (outbound-channels.md v1.3 "Redaction (enforced before emit)"; hosted-execution-
|
|
275
|
+
seams.md v1.11 §3 "any outbound projection ... redacts userinfo"). Two things, applied in
|
|
276
|
+
order:
|
|
277
|
+
|
|
278
|
+
1. URL userinfo (`_USERINFO_PATTERN`/`_mask_userinfo`, above) -- a password is masked and the
|
|
279
|
+
username kept (matching the contract's own `postgresql://harness:***@...` example); a bare
|
|
280
|
+
token/username-only userinfo (no `:`) is masked outright, since that shape is how a bearer
|
|
281
|
+
token appears, not a username.
|
|
282
|
+
2. `extra_secret_values` -- exact-substring replacement for a caller-supplied list of secret
|
|
283
|
+
values. Always `()` today: the adapter (P10) is what will know the job's declared secrets
|
|
284
|
+
and pass them in -- this parameter exists now so `build_event_record`/`build_result_receipt`
|
|
285
|
+
never need to change shape when that wiring lands.
|
|
286
|
+
|
|
287
|
+
NOT a general secret-content scanner -- the contract's "same secret-content scan as the
|
|
288
|
+
artifact sealer" is a separate, sealer-side obligation. This is the narrow subset this module
|
|
289
|
+
can enforce on every string field it controls without false-positiving on ordinary diagnostic
|
|
290
|
+
text.
|
|
291
|
+
"""
|
|
292
|
+
redacted = _USERINFO_PATTERN.sub(_mask_userinfo, value)
|
|
293
|
+
for secret in extra_secret_values:
|
|
294
|
+
if secret:
|
|
295
|
+
redacted = redacted.replace(secret, "***")
|
|
296
|
+
return redacted
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
# =================================================================================================
|
|
300
|
+
# Capabilities -- `/run/futureagi/capabilities.json`, "Authentication" section.
|
|
301
|
+
# =================================================================================================
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
class CapabilitiesError(RuntimeError):
|
|
305
|
+
"""A capability declaration is missing, malformed, or fails a shape rule.
|
|
306
|
+
|
|
307
|
+
Mirrors the `code`/`message` shape `BundleV2Error`/`PreflightError` use elsewhere in this
|
|
308
|
+
package. `code` is one of the nine closed values in outbound-channels.md v1.3's "Capabilities-
|
|
309
|
+
file rejection table" -- see `load_capabilities`, which is the sole place that maps a raw
|
|
310
|
+
failure onto one of them.
|
|
311
|
+
"""
|
|
312
|
+
|
|
313
|
+
def __init__(self, code: str, message: str) -> None:
|
|
314
|
+
self.code = code
|
|
315
|
+
self.message = message
|
|
316
|
+
super().__init__(f"{code}: {message}")
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
class HostedEndpoints(BaseModel):
|
|
320
|
+
"""The four outbound routes, per attempt. Shape-only checks (trailing slash, https) live here
|
|
321
|
+
as defense-in-depth for direct construction; `load_capabilities` runs the SAME checks earlier,
|
|
322
|
+
outside pydantic, so each has its own `CapabilitiesError.code` instead of collapsing into the
|
|
323
|
+
generic `capabilities_field_invalid` (see v1.3's rejection table)."""
|
|
324
|
+
|
|
325
|
+
model_config = ConfigDict(extra="forbid")
|
|
326
|
+
|
|
327
|
+
events: str
|
|
328
|
+
results: str
|
|
329
|
+
artifacts: str
|
|
330
|
+
scenarios: str
|
|
331
|
+
ingress: str | None = None
|
|
332
|
+
|
|
333
|
+
@model_validator(mode="after")
|
|
334
|
+
def _shape(self) -> "HostedEndpoints":
|
|
335
|
+
for name in ("events", "results", "artifacts", "scenarios", "ingress"):
|
|
336
|
+
value = getattr(self, name)
|
|
337
|
+
if value is None:
|
|
338
|
+
continue
|
|
339
|
+
if not value or not value.endswith("/"):
|
|
340
|
+
raise ValueError(f"{name} endpoint must end with '/'")
|
|
341
|
+
if not value.startswith("https://"):
|
|
342
|
+
raise ValueError(f"{name} endpoint must use https")
|
|
343
|
+
return self
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
class HostedCapabilities(BaseModel):
|
|
347
|
+
"""The per-attempt bearer plus the four endpoint URLs. Loaded once at emitter startup from the
|
|
348
|
+
file the gateway uploads (§0 step 4) -- see `load_capabilities`, which also implements the
|
|
349
|
+
contract's "loaded into memory ... and unlinked" lifetime rule."""
|
|
350
|
+
|
|
351
|
+
model_config = ConfigDict(extra="forbid")
|
|
352
|
+
|
|
353
|
+
schema_version: str
|
|
354
|
+
job_id: str = Field(min_length=1)
|
|
355
|
+
attempt_id: str = Field(min_length=1)
|
|
356
|
+
attempt_number: int = Field(ge=1)
|
|
357
|
+
fence: str = Field(min_length=1)
|
|
358
|
+
expires_at: UtcMillisDatetime
|
|
359
|
+
token: str = Field(min_length=1)
|
|
360
|
+
endpoints: HostedEndpoints
|
|
361
|
+
|
|
362
|
+
@model_validator(mode="after")
|
|
363
|
+
def _schema_shape(self) -> "HostedCapabilities":
|
|
364
|
+
# Defense-in-depth only -- `load_capabilities` checks this first, outside pydantic, so it
|
|
365
|
+
# can raise `capabilities_schema_unsupported` specifically rather than the generic
|
|
366
|
+
# `capabilities_field_invalid` this validator's `ValueError` would collapse into.
|
|
367
|
+
if self.schema_version != CAPABILITIES_SCHEMA_VERSION:
|
|
368
|
+
raise ValueError(f"unsupported schema_version: {self.schema_version}")
|
|
369
|
+
return self
|
|
370
|
+
|
|
371
|
+
@field_serializer("expires_at")
|
|
372
|
+
def _serialize_expires_at(self, value: datetime) -> str:
|
|
373
|
+
return format_rfc3339_millis(value)
|
|
374
|
+
|
|
375
|
+
def auth_headers(self) -> dict[str, str]:
|
|
376
|
+
""" "Every request: `Authorization: Bearer <token>` + `X-Harness-Fence: <fence>`." Pure
|
|
377
|
+
formatting -- issuing the request itself is a P8 transport-client concern."""
|
|
378
|
+
return {"Authorization": f"Bearer {self.token}", "X-Harness-Fence": self.fence}
|
|
379
|
+
|
|
380
|
+
def event_builder(
|
|
381
|
+
self, *, extra_secret_values: tuple[str, ...] = ()
|
|
382
|
+
) -> Callable[..., dict[str, Any]]:
|
|
383
|
+
"""A `build_event_record`-shaped callable with `job_id`/`attempt_id`/`attempt_number`
|
|
384
|
+
closed over from THIS capabilities object. The contract's `403 attempt_mismatch` fires when
|
|
385
|
+
an event's identity disagrees with the token authenticating it -- binding these three
|
|
386
|
+
fields here makes that class of caller bug unrepresentable at the call site instead of a
|
|
387
|
+
runtime 403 discovered mid-attempt.
|
|
388
|
+
|
|
389
|
+
P4: `extra_secret_values` is bound here too, alongside identity, rather than left as a
|
|
390
|
+
per-call parameter -- `build_event_record` could not previously receive the job's declared
|
|
391
|
+
secret list at all through this binder, so a caller wired for `event_builder()` had no way
|
|
392
|
+
to satisfy N9's redaction requirement on `log.message`/`terminal.failure.message` without
|
|
393
|
+
routing around this method entirely. Binding once here matches how identity is already
|
|
394
|
+
bound and gives the adapter one place to get it wrong instead of two.
|
|
395
|
+
"""
|
|
396
|
+
|
|
397
|
+
def build(
|
|
398
|
+
*,
|
|
399
|
+
event_id: str,
|
|
400
|
+
emitted_at: datetime,
|
|
401
|
+
stage: HarnessStage,
|
|
402
|
+
type: OutboundEventType,
|
|
403
|
+
payload: dict[str, JsonValue],
|
|
404
|
+
) -> dict[str, Any]:
|
|
405
|
+
return build_event_record(
|
|
406
|
+
event_id=event_id,
|
|
407
|
+
job_id=self.job_id,
|
|
408
|
+
attempt_id=self.attempt_id,
|
|
409
|
+
attempt_number=self.attempt_number,
|
|
410
|
+
emitted_at=emitted_at,
|
|
411
|
+
stage=stage,
|
|
412
|
+
type=type,
|
|
413
|
+
payload=payload,
|
|
414
|
+
extra_secret_values=extra_secret_values,
|
|
415
|
+
)
|
|
416
|
+
|
|
417
|
+
return build
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _endpoint_matches_attempt(url: str, attempt_id: str) -> bool:
|
|
421
|
+
""" "an endpoint's `<attempt_id>` path segment disagrees with the declared `attempt_id`"
|
|
422
|
+
(`capabilities_attempt_mismatch`, v1.3) -- checked as a whole path segment, not a substring, so
|
|
423
|
+
an attempt_id that happens to be a substring of another segment can't produce a false match."""
|
|
424
|
+
segments = [segment for segment in urlparse(url).path.split("/") if segment]
|
|
425
|
+
return attempt_id in segments
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _redact_validation_error(exc: ValidationError) -> str:
|
|
429
|
+
"""Builds a `capabilities_field_invalid` message from `loc`/`msg` only -- pydantic's default
|
|
430
|
+
`str(exc)` embeds each failing field's `input_value`, and the capabilities file carries the
|
|
431
|
+
bearer token; a caller that logs this message must never be able to leak it."""
|
|
432
|
+
parts = []
|
|
433
|
+
for error in exc.errors():
|
|
434
|
+
loc = ".".join(str(part) for part in error.get("loc", ()))
|
|
435
|
+
msg = error.get("msg", "")
|
|
436
|
+
parts.append(f"{loc}: {msg}" if loc else msg)
|
|
437
|
+
return "; ".join(parts) or "capabilities file failed validation"
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _warn_if_capabilities_file_insecure(target: Path) -> None:
|
|
441
|
+
""" "owner svc-control, mode 0600" is the contract's posture for this file, but a wrong mode or
|
|
442
|
+
owner is NOT a load-time rejection (MIN-5, fail-safe): the bearer is only a per-attempt token
|
|
443
|
+
that expires on its own, and refusing to load it entirely over a permissions mistake would turn
|
|
444
|
+
a minor hardening gap into a hard attempt failure. Loud warning only."""
|
|
445
|
+
try:
|
|
446
|
+
info = target.stat()
|
|
447
|
+
except OSError:
|
|
448
|
+
return
|
|
449
|
+
mode = info.st_mode & 0o777
|
|
450
|
+
if mode != 0o600:
|
|
451
|
+
logger.warning(
|
|
452
|
+
"%s: capabilities file mode is %o, expected 0600 -- a world/group-readable bearer in "
|
|
453
|
+
"a multi-user sandbox is a leak risk (not blocking the load)",
|
|
454
|
+
target,
|
|
455
|
+
mode,
|
|
456
|
+
)
|
|
457
|
+
try:
|
|
458
|
+
running_uid = os.geteuid()
|
|
459
|
+
except AttributeError:
|
|
460
|
+
return # os.geteuid() is POSIX-only
|
|
461
|
+
if info.st_uid != running_uid:
|
|
462
|
+
logger.warning(
|
|
463
|
+
"%s: capabilities file is owned by uid %s, not the running uid %s -- expected owner "
|
|
464
|
+
"svc-control per the contract (not blocking the load)",
|
|
465
|
+
target,
|
|
466
|
+
info.st_uid,
|
|
467
|
+
running_uid,
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def load_capabilities(
|
|
472
|
+
path: str | Path = CAPABILITIES_PATH,
|
|
473
|
+
*,
|
|
474
|
+
unlink: bool = True,
|
|
475
|
+
now: Callable[[], datetime] | None = None,
|
|
476
|
+
on_unlink_failure: Callable[[OSError], None] | None = None,
|
|
477
|
+
) -> HostedCapabilities:
|
|
478
|
+
"""Parse and validate one capabilities file against outbound-channels.md v1.3's closed
|
|
479
|
+
"Capabilities-file rejection table" (nine codes, all reachable as `CapabilitiesError.code`).
|
|
480
|
+
|
|
481
|
+
A guest that cannot load this file has no channel at all -- no token, no endpoints -- so it
|
|
482
|
+
cannot report its own failure; every branch below raises before any network-capable object
|
|
483
|
+
exists. Most checks run BEFORE `HostedCapabilities.model_validate`, outside any pydantic
|
|
484
|
+
validator: wrapping them in a pydantic `ValidationError` (the old shape) meant they only ever
|
|
485
|
+
surfaced as the generic `capabilities_field_invalid`, never as their own named code -- checking
|
|
486
|
+
here first makes each one an independently raised, independently testable `CapabilitiesError`.
|
|
487
|
+
|
|
488
|
+
``unlink=True`` (the default) implements "loaded into memory at emitter startup and unlinked" --
|
|
489
|
+
the file is only ever removed AFTER a successful parse and validation, never before, so a
|
|
490
|
+
crash mid-load leaves the file in place for the next attempt to read rather than destroying the
|
|
491
|
+
only copy of a not-yet-consumed bearer. A failure to unlink is never fatal to an otherwise-
|
|
492
|
+
successful load (the sandbox is destroyed by the gateway at attempt end regardless) but is no
|
|
493
|
+
longer silently swallowed either: it is reported via `on_unlink_failure` if given, else logged
|
|
494
|
+
(MIN-7) -- the caller can still tell a 0600 bearer may be lingering on disk.
|
|
495
|
+
|
|
496
|
+
``now`` is injectable (defaults to the real clock) so `capabilities_expired` is testable without
|
|
497
|
+
manipulating the wall clock.
|
|
498
|
+
"""
|
|
499
|
+
target = Path(path).expanduser()
|
|
500
|
+
try:
|
|
501
|
+
exists = target.is_file()
|
|
502
|
+
except OSError as exc:
|
|
503
|
+
raise CapabilitiesError("capabilities_file_unreadable", str(exc)) from exc
|
|
504
|
+
if not exists:
|
|
505
|
+
raise CapabilitiesError("capabilities_file_missing", str(target))
|
|
506
|
+
_warn_if_capabilities_file_insecure(target)
|
|
507
|
+
|
|
508
|
+
try:
|
|
509
|
+
text = target.read_text(encoding="utf-8")
|
|
510
|
+
except OSError as exc:
|
|
511
|
+
raise CapabilitiesError("capabilities_file_unreadable", str(exc)) from exc
|
|
512
|
+
try:
|
|
513
|
+
raw = json.loads(text)
|
|
514
|
+
except json.JSONDecodeError as exc:
|
|
515
|
+
raise CapabilitiesError("capabilities_file_malformed", str(exc)) from exc
|
|
516
|
+
if not isinstance(raw, dict):
|
|
517
|
+
raise CapabilitiesError("capabilities_file_malformed", "not a JSON object")
|
|
518
|
+
|
|
519
|
+
schema_version = raw.get("schema_version")
|
|
520
|
+
if schema_version != CAPABILITIES_SCHEMA_VERSION:
|
|
521
|
+
raise CapabilitiesError("capabilities_schema_unsupported", str(schema_version))
|
|
522
|
+
|
|
523
|
+
attempt_id = raw.get("attempt_id")
|
|
524
|
+
endpoints_raw = raw.get("endpoints")
|
|
525
|
+
if isinstance(endpoints_raw, dict):
|
|
526
|
+
for name in ("events", "results", "artifacts", "scenarios"):
|
|
527
|
+
value = endpoints_raw.get(name)
|
|
528
|
+
if not isinstance(value, str):
|
|
529
|
+
continue # missing/wrong-typed -- a shape error pydantic below will catch
|
|
530
|
+
if not value.endswith("/"):
|
|
531
|
+
raise CapabilitiesError(
|
|
532
|
+
"capabilities_endpoint_invalid",
|
|
533
|
+
f"endpoints.{name} must end with '/'",
|
|
534
|
+
)
|
|
535
|
+
if not value.startswith("https://"):
|
|
536
|
+
raise CapabilitiesError(
|
|
537
|
+
"capabilities_endpoint_insecure", f"endpoints.{name} must use https"
|
|
538
|
+
)
|
|
539
|
+
if (
|
|
540
|
+
isinstance(attempt_id, str)
|
|
541
|
+
and attempt_id
|
|
542
|
+
and not _endpoint_matches_attempt(value, attempt_id)
|
|
543
|
+
):
|
|
544
|
+
raise CapabilitiesError(
|
|
545
|
+
"capabilities_attempt_mismatch",
|
|
546
|
+
f"endpoints.{name} does not carry the declared attempt_id {attempt_id!r}",
|
|
547
|
+
)
|
|
548
|
+
|
|
549
|
+
try:
|
|
550
|
+
capabilities = HostedCapabilities.model_validate(raw)
|
|
551
|
+
except ValidationError as exc:
|
|
552
|
+
raise CapabilitiesError(
|
|
553
|
+
"capabilities_field_invalid", _redact_validation_error(exc)
|
|
554
|
+
) from exc
|
|
555
|
+
|
|
556
|
+
current_time = (now or (lambda: datetime.now(timezone.utc)))()
|
|
557
|
+
if capabilities.expires_at <= current_time:
|
|
558
|
+
raise CapabilitiesError(
|
|
559
|
+
"capabilities_expired", f"expires_at={capabilities.expires_at.isoformat()}"
|
|
560
|
+
)
|
|
561
|
+
|
|
562
|
+
if unlink:
|
|
563
|
+
try:
|
|
564
|
+
target.unlink()
|
|
565
|
+
except OSError as exc:
|
|
566
|
+
if on_unlink_failure is not None:
|
|
567
|
+
on_unlink_failure(exc)
|
|
568
|
+
else:
|
|
569
|
+
logger.warning(
|
|
570
|
+
"%s: failed to unlink the capabilities file after a successful load (%s) -- a "
|
|
571
|
+
"0600 bearer may still be on disk; the sandbox is destroyed at attempt end "
|
|
572
|
+
"regardless, so this does not fail the load",
|
|
573
|
+
target,
|
|
574
|
+
exc,
|
|
575
|
+
)
|
|
576
|
+
return capabilities
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
# =================================================================================================
|
|
580
|
+
# Channel 1 -- Events. The closed `type` vocabulary and each type's payload shape.
|
|
581
|
+
# =================================================================================================
|
|
582
|
+
|
|
583
|
+
|
|
584
|
+
class DegradeReason(str, Enum):
|
|
585
|
+
CONFORMANCE_GATE_FAILED = "conformance_gate_failed"
|
|
586
|
+
FIXED_PORT = "fixed_port"
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
class LogLevel(str, Enum):
|
|
590
|
+
DEBUG = "debug"
|
|
591
|
+
INFO = "info"
|
|
592
|
+
WARNING = "warning"
|
|
593
|
+
ERROR = "error"
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
class TerminalReason(str, Enum):
|
|
597
|
+
TTL_EXCEEDED = "ttl_exceeded"
|
|
598
|
+
USER_CANCELED = "user_canceled"
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
class OutboundEventType(str, Enum):
|
|
602
|
+
STAGE_CHANGED = "stage_changed"
|
|
603
|
+
PARALLELISM_DEGRADED = "parallelism_degraded"
|
|
604
|
+
BASELINE_FROZEN = "baseline_frozen"
|
|
605
|
+
BASELINE_INPUTS_CHANGED = "baseline_inputs_changed"
|
|
606
|
+
WORLD_UNHEALTHY = "world_unhealthy"
|
|
607
|
+
SCENARIO_STARTED = "scenario_started"
|
|
608
|
+
SCENARIO_RETRIED = "scenario_retried"
|
|
609
|
+
LOG = "log"
|
|
610
|
+
TERMINAL = "terminal"
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
class StageChangedPayload(BaseModel):
|
|
614
|
+
"""No `populate_by_name` -- the wire key is `from` (a Python keyword, hence the `from_stage`
|
|
615
|
+
attribute name + alias), and this model is validation-only (`HostedEventDraft` never
|
|
616
|
+
normalizes `self.payload`; it emits the caller's dict verbatim). Allowing population by the
|
|
617
|
+
attribute name too would let a caller who writes `from_stage` in their payload dict pass
|
|
618
|
+
validation while spooling an undefined wire key -- `populate_by_name=True` previously made
|
|
619
|
+
exactly that mistake succeed silently."""
|
|
620
|
+
|
|
621
|
+
model_config = ConfigDict(extra="forbid")
|
|
622
|
+
|
|
623
|
+
from_stage: HarnessStage | None = Field(alias="from")
|
|
624
|
+
to: HarnessStage
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
class ParallelismDegradedPayload(BaseModel):
|
|
628
|
+
model_config = ConfigDict(extra="forbid")
|
|
629
|
+
|
|
630
|
+
requested: int = Field(ge=1)
|
|
631
|
+
effective: int = Field(ge=1)
|
|
632
|
+
reason: DegradeReason
|
|
633
|
+
|
|
634
|
+
@model_validator(mode="after")
|
|
635
|
+
def _range(self) -> "ParallelismDegradedPayload":
|
|
636
|
+
if not (1 <= self.effective < self.requested):
|
|
637
|
+
raise ValueError(
|
|
638
|
+
f"parallelism_degraded_effective_out_of_range: effective={self.effective} "
|
|
639
|
+
f"requested={self.requested}"
|
|
640
|
+
)
|
|
641
|
+
return self
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
class BaselineFrozenPayload(BaseModel):
|
|
645
|
+
model_config = ConfigDict(extra="forbid")
|
|
646
|
+
|
|
647
|
+
inputs_digest: str
|
|
648
|
+
baseline_ref: str
|
|
649
|
+
|
|
650
|
+
|
|
651
|
+
class BaselineInputsChangedPayload(BaseModel):
|
|
652
|
+
model_config = ConfigDict(extra="forbid")
|
|
653
|
+
|
|
654
|
+
previous_digest: str | None
|
|
655
|
+
current_digest: str
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
class WorldUnhealthyPayload(BaseModel):
|
|
659
|
+
model_config = ConfigDict(extra="forbid")
|
|
660
|
+
|
|
661
|
+
world_index: int = Field(ge=0)
|
|
662
|
+
cause: str = Field(max_length=200)
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
class ScenarioStartedPayload(BaseModel):
|
|
666
|
+
model_config = ConfigDict(extra="forbid")
|
|
667
|
+
|
|
668
|
+
scenario_key: str = Field(min_length=1)
|
|
669
|
+
world_index: int = Field(ge=0)
|
|
670
|
+
scenario_attempt: Literal[1, 2]
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
class ScenarioRetriedPayload(BaseModel):
|
|
674
|
+
model_config = ConfigDict(extra="forbid")
|
|
675
|
+
|
|
676
|
+
scenario_key: str = Field(min_length=1)
|
|
677
|
+
from_world: int = Field(ge=0)
|
|
678
|
+
to_world: int = Field(ge=0)
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
class LogPayload(BaseModel):
|
|
682
|
+
model_config = ConfigDict(extra="forbid")
|
|
683
|
+
|
|
684
|
+
level: LogLevel
|
|
685
|
+
message: str
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
_LOG_TRUNCATION_MARKER = "…[truncated]"
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
def truncate_log_message(
|
|
692
|
+
level: str, message: str, *, max_payload_bytes: int = EVENT_PAYLOAD_MAX_BYTES
|
|
693
|
+
) -> str:
|
|
694
|
+
"""The `log` event's own contract rule -- "truncated to fit with a trailing `…[truncated]`
|
|
695
|
+
marker" -- unlike the other eight event types, which are hard-rejected when oversized (M8:
|
|
696
|
+
`log` is the contract's designated escape hatch for reporting every other permanent failure, so
|
|
697
|
+
the one channel meant to report an oversized diagnostic must not itself throw on size).
|
|
698
|
+
|
|
699
|
+
Sizing is against `canonical_bytes({"level": level, "message": <candidate>})`, the exact bytes
|
|
700
|
+
`HostedEventDraft`'s own size check measures, so a truncated message is guaranteed to fit
|
|
701
|
+
before it ever reaches that check. A no-op when already within budget.
|
|
702
|
+
"""
|
|
703
|
+
if len(canonical_bytes({"level": level, "message": message})) <= max_payload_bytes:
|
|
704
|
+
return message
|
|
705
|
+
if (
|
|
706
|
+
len(canonical_bytes({"level": level, "message": _LOG_TRUNCATION_MARKER}))
|
|
707
|
+
> max_payload_bytes
|
|
708
|
+
):
|
|
709
|
+
raise OutboundError(
|
|
710
|
+
"log_payload_budget_too_small",
|
|
711
|
+
f"max_payload_bytes={max_payload_bytes} cannot fit even the truncation marker",
|
|
712
|
+
)
|
|
713
|
+
lo, hi, best = 0, len(message), ""
|
|
714
|
+
while lo <= hi:
|
|
715
|
+
mid = (lo + hi) // 2
|
|
716
|
+
candidate = message[:mid] + _LOG_TRUNCATION_MARKER
|
|
717
|
+
if (
|
|
718
|
+
len(canonical_bytes({"level": level, "message": candidate}))
|
|
719
|
+
<= max_payload_bytes
|
|
720
|
+
):
|
|
721
|
+
best = candidate
|
|
722
|
+
lo = mid + 1
|
|
723
|
+
else:
|
|
724
|
+
hi = mid - 1
|
|
725
|
+
return best
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
class TerminalFailure(BaseModel):
|
|
729
|
+
"""The terminal event's `failure` shape: `{domain, stage, code, message}` -- a leaner subset of
|
|
730
|
+
`job.HarnessFailure` (no `retryable`/`details`), matching §"Event `type` vocabulary" exactly so
|
|
731
|
+
a canonicalized terminal payload never carries fields the contract doesn't name."""
|
|
732
|
+
|
|
733
|
+
model_config = ConfigDict(extra="forbid")
|
|
734
|
+
|
|
735
|
+
domain: FailureDomain
|
|
736
|
+
stage: HarnessStage
|
|
737
|
+
code: str
|
|
738
|
+
message: str
|
|
739
|
+
|
|
740
|
+
|
|
741
|
+
class ScenarioCounts(BaseModel):
|
|
742
|
+
model_config = ConfigDict(extra="forbid")
|
|
743
|
+
|
|
744
|
+
passed: int = Field(ge=0)
|
|
745
|
+
failed: int = Field(ge=0)
|
|
746
|
+
errored: int = Field(ge=0)
|
|
747
|
+
skipped: int = Field(ge=0)
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
class TerminalPayload(BaseModel):
|
|
751
|
+
model_config = ConfigDict(extra="forbid")
|
|
752
|
+
|
|
753
|
+
stage: HarnessStage
|
|
754
|
+
reason: TerminalReason | None
|
|
755
|
+
failure: TerminalFailure | None
|
|
756
|
+
scenario_counts: ScenarioCounts
|
|
757
|
+
|
|
758
|
+
@model_validator(mode="after")
|
|
759
|
+
def _terminal_stage(self) -> "TerminalPayload":
|
|
760
|
+
if not self.stage.terminal:
|
|
761
|
+
raise ValueError(f"terminal_event_stage_not_terminal: {self.stage.value}")
|
|
762
|
+
return self
|
|
763
|
+
|
|
764
|
+
|
|
765
|
+
_PAYLOAD_MODELS: dict[OutboundEventType, type[BaseModel]] = {
|
|
766
|
+
OutboundEventType.STAGE_CHANGED: StageChangedPayload,
|
|
767
|
+
OutboundEventType.PARALLELISM_DEGRADED: ParallelismDegradedPayload,
|
|
768
|
+
OutboundEventType.BASELINE_FROZEN: BaselineFrozenPayload,
|
|
769
|
+
OutboundEventType.BASELINE_INPUTS_CHANGED: BaselineInputsChangedPayload,
|
|
770
|
+
OutboundEventType.WORLD_UNHEALTHY: WorldUnhealthyPayload,
|
|
771
|
+
OutboundEventType.SCENARIO_STARTED: ScenarioStartedPayload,
|
|
772
|
+
OutboundEventType.SCENARIO_RETRIED: ScenarioRetriedPayload,
|
|
773
|
+
OutboundEventType.LOG: LogPayload,
|
|
774
|
+
OutboundEventType.TERMINAL: TerminalPayload,
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
class HostedEventDraft(BaseModel):
|
|
779
|
+
"""A Channel 1 event before spool-assigned `sequence`. The caller supplies `digest` itself
|
|
780
|
+
(computed via `event_payload_digest`) -- the model then re-derives it and rejects a mismatch,
|
|
781
|
+
so a caller can never accidentally spool a record whose embedded digest disagrees with its own
|
|
782
|
+
payload bytes.
|
|
783
|
+
"""
|
|
784
|
+
|
|
785
|
+
model_config = ConfigDict(extra="forbid")
|
|
786
|
+
|
|
787
|
+
event_id: str = Field(min_length=1, max_length=64)
|
|
788
|
+
job_id: str = Field(min_length=1)
|
|
789
|
+
attempt_id: str = Field(min_length=1)
|
|
790
|
+
attempt_number: int = Field(ge=1)
|
|
791
|
+
emitted_at: UtcMillisDatetime
|
|
792
|
+
stage: HarnessStage
|
|
793
|
+
type: OutboundEventType
|
|
794
|
+
payload: dict[str, JsonValue]
|
|
795
|
+
digest: str
|
|
796
|
+
|
|
797
|
+
@field_serializer("emitted_at")
|
|
798
|
+
def _serialize_emitted_at(self, value: datetime) -> str:
|
|
799
|
+
return format_rfc3339_millis(value)
|
|
800
|
+
|
|
801
|
+
@model_validator(mode="after")
|
|
802
|
+
def _validate(self) -> "HostedEventDraft":
|
|
803
|
+
# `max_length=64` above counts characters; the contract says "opaque <=64 chars," but the
|
|
804
|
+
# platform's column is presumably bytes -- a multi-byte-UTF-8 id could pass the character
|
|
805
|
+
# count and still overflow it (N-2).
|
|
806
|
+
if len(self.event_id.encode("utf-8")) > 64:
|
|
807
|
+
raise ValueError(f"event_id_too_long_in_bytes: {self.event_id!r}")
|
|
808
|
+
if not is_valid_digest(self.digest):
|
|
809
|
+
raise ValueError(f"event_digest_invalid: {self.digest!r}")
|
|
810
|
+
expected = event_payload_digest(self.payload)
|
|
811
|
+
if self.digest != expected:
|
|
812
|
+
raise ValueError("event_digest_mismatch")
|
|
813
|
+
if len(canonical_bytes(self.payload)) > EVENT_PAYLOAD_MAX_BYTES:
|
|
814
|
+
raise ValueError(f"event_payload_too_large: {self.event_id}")
|
|
815
|
+
|
|
816
|
+
model_cls = _PAYLOAD_MODELS[self.type]
|
|
817
|
+
try:
|
|
818
|
+
model_cls.model_validate(self.payload)
|
|
819
|
+
except ValidationError as exc:
|
|
820
|
+
raise ValueError(
|
|
821
|
+
f"event_payload_invalid: {self.type.value}: {exc}"
|
|
822
|
+
) from exc
|
|
823
|
+
|
|
824
|
+
if (
|
|
825
|
+
self.type is OutboundEventType.STAGE_CHANGED
|
|
826
|
+
and self.payload.get("to") != self.stage.value
|
|
827
|
+
):
|
|
828
|
+
raise ValueError(
|
|
829
|
+
"event_stage_mismatch: stage_changed.to must equal the event's stage"
|
|
830
|
+
)
|
|
831
|
+
if (
|
|
832
|
+
self.type is OutboundEventType.TERMINAL
|
|
833
|
+
and self.payload.get("stage") != self.stage.value
|
|
834
|
+
):
|
|
835
|
+
raise ValueError(
|
|
836
|
+
"event_stage_mismatch: terminal.stage must equal the event's stage"
|
|
837
|
+
)
|
|
838
|
+
return self
|
|
839
|
+
|
|
840
|
+
|
|
841
|
+
class HostedEvent(HostedEventDraft):
|
|
842
|
+
"""The full Channel 1 wire object, `sequence` included -- what actually gets spooled and sent.
|
|
843
|
+
Distinct from `fi.simulate.runtime.events.CanonicalEvent` (the untouched local-SDK wire)."""
|
|
844
|
+
|
|
845
|
+
sequence: int = Field(ge=1)
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
def build_event_record(
|
|
849
|
+
*,
|
|
850
|
+
event_id: str,
|
|
851
|
+
job_id: str,
|
|
852
|
+
attempt_id: str,
|
|
853
|
+
attempt_number: int,
|
|
854
|
+
emitted_at: datetime,
|
|
855
|
+
stage: HarnessStage,
|
|
856
|
+
type: OutboundEventType,
|
|
857
|
+
payload: dict[str, JsonValue],
|
|
858
|
+
extra_secret_values: tuple[str, ...] = (),
|
|
859
|
+
) -> dict[str, Any]:
|
|
860
|
+
"""Validate one event's shape and compute its digest, returning a plain dict with no
|
|
861
|
+
`sequence` key -- ready for `OutboundSpool.append`, which assigns `sequence` and performs the
|
|
862
|
+
one-time serialization. Raises `ValueError` (via pydantic) on any shape violation; callers that
|
|
863
|
+
want a typed/coded failure should catch `pydantic.ValidationError` themselves, matching how the
|
|
864
|
+
rest of this package surfaces model-layer rejections (`bundle_v2.py`, `job.py`).
|
|
865
|
+
|
|
866
|
+
N9: `redact_outbound_text` runs on every free-text field the contract names BEFORE the digest
|
|
867
|
+
is computed -- `log.message`, `world_unhealthy.cause`, `terminal.failure.{code,message}`,
|
|
868
|
+
`baseline_frozen.baseline_ref` (P8) -- so the embedded digest always matches the redacted bytes
|
|
869
|
+
actually spooled and sent, never the unredacted original. `log` events are then truncated to
|
|
870
|
+
fit (M8), also before the digest -- redact first, since truncation must size against the final
|
|
871
|
+
(redacted) text, not text that would still shrink again once secrets are scrubbed. Every other
|
|
872
|
+
event type still hard-rejects when oversized, via `HostedEventDraft`'s own size check.
|
|
873
|
+
|
|
874
|
+
P8: `failure.code` is redacted alongside `failure.message` -- both are free `str` fields (the
|
|
875
|
+
contract's `code` vocabularies are closed in prose, but nothing enforces that here), and
|
|
876
|
+
`baseline_ref` is likewise a free `str` that plausibly carries an OCI/registry reference in the
|
|
877
|
+
same `https://<token>@registry/...` shape `redact_outbound_text` already scrubs.
|
|
878
|
+
"""
|
|
879
|
+
payload = dict(payload)
|
|
880
|
+
if type is OutboundEventType.LOG:
|
|
881
|
+
level, message = payload.get("level"), payload.get("message")
|
|
882
|
+
if isinstance(message, str):
|
|
883
|
+
message = redact_outbound_text(message, extra_secret_values)
|
|
884
|
+
if isinstance(level, str):
|
|
885
|
+
message = truncate_log_message(level, message)
|
|
886
|
+
payload["message"] = message
|
|
887
|
+
elif type is OutboundEventType.WORLD_UNHEALTHY:
|
|
888
|
+
cause = payload.get("cause")
|
|
889
|
+
if isinstance(cause, str):
|
|
890
|
+
payload["cause"] = redact_outbound_text(cause, extra_secret_values)
|
|
891
|
+
elif type is OutboundEventType.BASELINE_FROZEN:
|
|
892
|
+
baseline_ref = payload.get("baseline_ref")
|
|
893
|
+
if isinstance(baseline_ref, str):
|
|
894
|
+
payload["baseline_ref"] = redact_outbound_text(
|
|
895
|
+
baseline_ref, extra_secret_values
|
|
896
|
+
)
|
|
897
|
+
elif type is OutboundEventType.TERMINAL:
|
|
898
|
+
failure = payload.get("failure")
|
|
899
|
+
if isinstance(failure, dict):
|
|
900
|
+
redacted_failure = dict(failure)
|
|
901
|
+
if isinstance(failure.get("code"), str):
|
|
902
|
+
redacted_failure["code"] = redact_outbound_text(
|
|
903
|
+
failure["code"], extra_secret_values
|
|
904
|
+
)
|
|
905
|
+
if isinstance(failure.get("message"), str):
|
|
906
|
+
redacted_failure["message"] = redact_outbound_text(
|
|
907
|
+
failure["message"], extra_secret_values
|
|
908
|
+
)
|
|
909
|
+
payload["failure"] = redacted_failure
|
|
910
|
+
digest = event_payload_digest(payload)
|
|
911
|
+
draft = HostedEventDraft(
|
|
912
|
+
event_id=event_id,
|
|
913
|
+
job_id=job_id,
|
|
914
|
+
attempt_id=attempt_id,
|
|
915
|
+
attempt_number=attempt_number,
|
|
916
|
+
emitted_at=emitted_at,
|
|
917
|
+
stage=stage,
|
|
918
|
+
type=type,
|
|
919
|
+
payload=payload,
|
|
920
|
+
digest=digest,
|
|
921
|
+
)
|
|
922
|
+
return draft.model_dump(mode="json")
|
|
923
|
+
|
|
924
|
+
|
|
925
|
+
# =================================================================================================
|
|
926
|
+
# Spool -- durable on-disk queue + monotonic sequence allocator.
|
|
927
|
+
# =================================================================================================
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
@dataclass(frozen=True)
|
|
931
|
+
class SpooledRecord:
|
|
932
|
+
"""One durably-appended record. `body` is the EXACT canonical bytes written to disk -- a P8
|
|
933
|
+
transport client re-sends `body` verbatim on retry rather than re-serializing the decoded
|
|
934
|
+
dict, per the contract's "serialize once ... never re-serialize on retry.\""""
|
|
935
|
+
|
|
936
|
+
sequence: int | None
|
|
937
|
+
body: bytes
|
|
938
|
+
|
|
939
|
+
def decode(self) -> dict[str, Any]:
|
|
940
|
+
return json.loads(self.body.decode("utf-8"))
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
class OutboundSpoolError(RuntimeError):
|
|
944
|
+
def __init__(self, code: str, message: str) -> None:
|
|
945
|
+
self.code = code
|
|
946
|
+
self.message = message
|
|
947
|
+
super().__init__(f"{code}: {message}")
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def _iter_complete_records(
|
|
951
|
+
data: bytes,
|
|
952
|
+
) -> tuple[
|
|
953
|
+
list[
|
|
954
|
+
tuple[int, bytes, dict[str, Any] | list[Any] | str | int | float | bool | None]
|
|
955
|
+
],
|
|
956
|
+
int,
|
|
957
|
+
int | None,
|
|
958
|
+
]:
|
|
959
|
+
"""Shared by `_recover` and `records()`/`pending_since_watermark()` -- the ONE place spool
|
|
960
|
+
bytes are split into records, so both ever agree on what a "complete record" is.
|
|
961
|
+
|
|
962
|
+
Splits on the literal `b"\\n"` byte (N-1: not `bytes.splitlines()`, which also treats `\\r`/
|
|
963
|
+
`\\r\\n` as separators `append` never writes -- canonical JSON never contains a raw newline of
|
|
964
|
+
any kind, so `\\n` is the only byte that can legitimately end a line).
|
|
965
|
+
|
|
966
|
+
Returns `(records_before_corruption, valid_length, corruption_offset)` (N8):
|
|
967
|
+
- `records_before_corruption`: every complete, successfully decoded, non-blank line UP TO the
|
|
968
|
+
first corrupt one (or all of them, if none is corrupt), as `(start_offset, raw_line,
|
|
969
|
+
decoded_value)`, in file order.
|
|
970
|
+
- `corruption_offset`: the byte offset of the first `\\n`-terminated line that failed to parse
|
|
971
|
+
as JSON -- genuine corruption (a torn write, by definition, never got its trailing `\\n`, so
|
|
972
|
+
this is never that) -- or `None` if no such line was found. Once found, scanning STOPS: never
|
|
973
|
+
renumber or trust anything past a corrupt byte (B1).
|
|
974
|
+
- `valid_length`: when `corruption_offset is None`, the byte offset immediately after the last
|
|
975
|
+
complete line -- where `_recover` truncates away a torn tail. When corruption WAS found, this
|
|
976
|
+
equals `corruption_offset` and callers must NOT use it to truncate -- corrupt bytes are left
|
|
977
|
+
on disk, never deleted (N8: "never truncates mid-file damage").
|
|
978
|
+
"""
|
|
979
|
+
records: list[tuple[int, bytes, Any]] = []
|
|
980
|
+
start = 0
|
|
981
|
+
while True:
|
|
982
|
+
newline_index = data.find(b"\n", start)
|
|
983
|
+
if newline_index == -1:
|
|
984
|
+
return records, start, None
|
|
985
|
+
line = data[start:newline_index]
|
|
986
|
+
line_end = newline_index + 1
|
|
987
|
+
if line:
|
|
988
|
+
try:
|
|
989
|
+
decoded = json.loads(line.decode("utf-8"))
|
|
990
|
+
except ValueError:
|
|
991
|
+
return records, start, start
|
|
992
|
+
records.append((start, line, decoded))
|
|
993
|
+
start = line_end
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
class OutboundSpool:
|
|
997
|
+
"""Durable, crash-safe local queue for one outbound record stream (events, results, or
|
|
998
|
+
artifact-manifest state) -- the "fsync-first local spool" the contract requires emission to sit
|
|
999
|
+
behind ("Emission is an async flusher over the fsync-first local spool -- it never blocks the
|
|
1000
|
+
call loop"). `sequenced=True` is for Channel 1 only ("Sequencing: one allocator, one lock,
|
|
1001
|
+
assigned at spool append, contiguous from 1" -- receipts and the manifest carry no `sequence`
|
|
1002
|
+
field and use their own idempotency keys instead).
|
|
1003
|
+
|
|
1004
|
+
ONE ALLOCATOR PER STREAM (M6): `OutboundSpool(root, name, ...)` is keyed on
|
|
1005
|
+
`(resolved_root, name)` -- a second construction for the same key, anywhere in this process,
|
|
1006
|
+
returns the SAME instance rather than a second independent allocator (see `__new__`); a second
|
|
1007
|
+
OS PROCESS pointing at the same directory fails loudly instead, via an `fcntl.flock` on
|
|
1008
|
+
`<name>.spool.lock` held for the life of the owning instance.
|
|
1009
|
+
|
|
1010
|
+
Recovery rule (the contract specifies the sequencing invariant -- contiguous from 1, no gaps or
|
|
1011
|
+
dupes across a restart -- but not the recovery mechanism; this is the FAIL-SAFE/REVERSIBLE
|
|
1012
|
+
choice under the stuck-decision rule, surfaced in the P7 report):
|
|
1013
|
+
|
|
1014
|
+
The next sequence number is derived by SCANNING the spool's own JSONL log at startup, never
|
|
1015
|
+
from an independent counter file. A separate counter file could be durably advanced in a write
|
|
1016
|
+
that lands, while the record it was allocated for does not (crash between the two writes),
|
|
1017
|
+
producing a sequence number with no corresponding record -- a permanent, undetectable gap.
|
|
1018
|
+
Scanning the log makes the durably-written records themselves the only source of truth. The
|
|
1019
|
+
scan is then reconciled against the durable watermark (M5): `next_sequence =
|
|
1020
|
+
max(max_sequence_in_log, watermark) + 1` -- the watermark can be AHEAD of the log (the log lost
|
|
1021
|
+
already-processed records, e.g. via the directory-fsync gap M2 closes) but never behind it, so
|
|
1022
|
+
taking the max is always safe and never skips a record that was actually spooled.
|
|
1023
|
+
|
|
1024
|
+
A torn last line -- a crash mid-write, since a single `write()` of `body + b"\\n"` is not
|
|
1025
|
+
guaranteed atomic by POSIX for a regular file -- is detected (the trailing bytes don't end in
|
|
1026
|
+
`b"\\n"`) and the file is truncated back to the end of the last complete record before any
|
|
1027
|
+
further append. The next append then reuses that same sequence number rather than skipping it:
|
|
1028
|
+
a torn write is treated as though it never happened, closing the gap instead of creating one.
|
|
1029
|
+
This depends on canonical JSON never containing a raw newline byte (control characters are
|
|
1030
|
+
always escaped by `json.dumps`), which `append` asserts on every write. A COMPLETE line that
|
|
1031
|
+
still fails to parse is a different, worse fault: genuine corruption degrades the stream to its
|
|
1032
|
+
readable prefix rather than raising (N8, `is_corrupt`) -- see `_recover`/`records()`.
|
|
1033
|
+
|
|
1034
|
+
Registration (N4): a construction is only added to `_registry` at the END of a successful
|
|
1035
|
+
`__init__`, under `_registry_lock` -- never a half-built instance. A failed construction (e.g.
|
|
1036
|
+
`mkdir` EACCES, or `_recover` finding corruption) therefore never poisons the key: it raises
|
|
1037
|
+
without registering anything, and the NEXT `OutboundSpool(root, name, ...)` call starts a
|
|
1038
|
+
completely fresh attempt rather than returning (or conflicting with) wreckage. The real mutual-
|
|
1039
|
+
exclusion primitive across a same-key construction race is `_acquire_process_lock`'s `flock`
|
|
1040
|
+
(an OS-level device, safe across threads and processes alike) -- the registry dict on top is
|
|
1041
|
+
only a same-process memoization cache.
|
|
1042
|
+
"""
|
|
1043
|
+
|
|
1044
|
+
_registry: ClassVar[dict[tuple[Path, str], "OutboundSpool"]] = {}
|
|
1045
|
+
_registry_lock: ClassVar[RLock] = RLock()
|
|
1046
|
+
|
|
1047
|
+
def __new__(
|
|
1048
|
+
cls, root: str | Path, name: str, *, sequenced: bool
|
|
1049
|
+
) -> "OutboundSpool":
|
|
1050
|
+
resolved_root = Path(root).expanduser().resolve()
|
|
1051
|
+
key = (resolved_root, name)
|
|
1052
|
+
with cls._registry_lock:
|
|
1053
|
+
existing = cls._registry.get(key)
|
|
1054
|
+
if existing is not None:
|
|
1055
|
+
# getattr belt-and-braces (N4): `existing` is only ever registered after a fully
|
|
1056
|
+
# successful __init__, so `_sequenced` should always be set -- but never trust that
|
|
1057
|
+
# invariant harder than a defensive read costs.
|
|
1058
|
+
if getattr(existing, "_sequenced", None) != sequenced:
|
|
1059
|
+
raise OutboundSpoolError(
|
|
1060
|
+
"outbound_spool_sequenced_mismatch",
|
|
1061
|
+
f"{name}: existing instance has sequenced={getattr(existing, '_sequenced', None)}, "
|
|
1062
|
+
f"requested sequenced={sequenced}",
|
|
1063
|
+
)
|
|
1064
|
+
return existing
|
|
1065
|
+
return super().__new__(cls)
|
|
1066
|
+
|
|
1067
|
+
def __init__(self, root: str | Path, name: str, *, sequenced: bool) -> None:
|
|
1068
|
+
if getattr(self, "_initialized", False):
|
|
1069
|
+
return
|
|
1070
|
+
resolved_root = Path(root).expanduser().resolve()
|
|
1071
|
+
key = (resolved_root, name)
|
|
1072
|
+
with type(self)._registry_lock:
|
|
1073
|
+
if getattr(self, "_initialized", False):
|
|
1074
|
+
return
|
|
1075
|
+
lock_fd: int | None = None
|
|
1076
|
+
try:
|
|
1077
|
+
self.root = resolved_root
|
|
1078
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
1079
|
+
try:
|
|
1080
|
+
os.chmod(
|
|
1081
|
+
self.root, 0o700
|
|
1082
|
+
) # MIN-10: mkdir's mode is subject to umask
|
|
1083
|
+
except OSError:
|
|
1084
|
+
pass
|
|
1085
|
+
self._name = name
|
|
1086
|
+
self._sequenced = sequenced
|
|
1087
|
+
self._path = self.root / f"{name}.spool.jsonl"
|
|
1088
|
+
self._watermark_path = self.root / f"{name}.spool.watermark.json"
|
|
1089
|
+
self._lock = RLock()
|
|
1090
|
+
self._dir_synced = False
|
|
1091
|
+
self._offset_by_sequence: dict[int, int] = {}
|
|
1092
|
+
self._next_sequence = 1 if sequenced else None
|
|
1093
|
+
self._poisoned = False # N12
|
|
1094
|
+
self._corrupt_since_offset: int | None = None # N8
|
|
1095
|
+
self._forked = False # N25
|
|
1096
|
+
self._closed = False # P2
|
|
1097
|
+
self._lock_fd = self._acquire_process_lock()
|
|
1098
|
+
lock_fd = self._lock_fd
|
|
1099
|
+
self._recover()
|
|
1100
|
+
except BaseException:
|
|
1101
|
+
# N4: never leave a half-built instance registered -- it was never added (below),
|
|
1102
|
+
# so there is nothing to evict; just release whatever this attempt itself opened.
|
|
1103
|
+
if lock_fd is not None:
|
|
1104
|
+
try:
|
|
1105
|
+
os.close(lock_fd)
|
|
1106
|
+
except OSError:
|
|
1107
|
+
pass
|
|
1108
|
+
raise
|
|
1109
|
+
self._initialized = True
|
|
1110
|
+
type(self)._registry[key] = self
|
|
1111
|
+
|
|
1112
|
+
@classmethod
|
|
1113
|
+
def _forget_for_tests(cls, root: str | Path, name: str) -> None:
|
|
1114
|
+
"""Test-only escape hatch: a real process restart naturally gets a fresh, empty registry
|
|
1115
|
+
(a new interpreter); simulating that WITHIN one process/test needs an explicit evict so the
|
|
1116
|
+
next `OutboundSpool(root, name, ...)` call re-scans the on-disk log instead of returning the
|
|
1117
|
+
still-live cached instance. Never called from production code."""
|
|
1118
|
+
resolved_root = Path(root).expanduser().resolve()
|
|
1119
|
+
with cls._registry_lock:
|
|
1120
|
+
instance = cls._registry.get((resolved_root, name))
|
|
1121
|
+
if instance is not None:
|
|
1122
|
+
instance.close()
|
|
1123
|
+
|
|
1124
|
+
@classmethod
|
|
1125
|
+
def _clear_registry_for_tests(cls) -> None:
|
|
1126
|
+
"""Broader sibling of `_forget_for_tests`: releases every cached instance's lock fd and
|
|
1127
|
+
empties the registry. Intended for an autouse test fixture so flock fds don't accumulate
|
|
1128
|
+
across a whole test session."""
|
|
1129
|
+
with cls._registry_lock:
|
|
1130
|
+
instances = list(cls._registry.values())
|
|
1131
|
+
for instance in instances:
|
|
1132
|
+
instance.close()
|
|
1133
|
+
|
|
1134
|
+
def close(self) -> None:
|
|
1135
|
+
"""N26/P2: releases this instance's process lock and evicts it from the registry, so a
|
|
1136
|
+
later `OutboundSpool(root, name, ...)` call re-scans the on-disk log instead of reusing this
|
|
1137
|
+
instance. Idempotent -- safe to call more than once, or on an instance never fully
|
|
1138
|
+
constructed.
|
|
1139
|
+
|
|
1140
|
+
P2: also sets `_closed`, so THIS instance -- not just the registry slot -- refuses further
|
|
1141
|
+
mutation. Evicting the registry entry alone left the closed instance itself fully live: a
|
|
1142
|
+
caller still holding a reference could keep appending with no flock held (the lock fd was
|
|
1143
|
+
released), and a fresh `OutboundSpool(...)` call for the same key would allocate a second,
|
|
1144
|
+
independent `_next_sequence` -- two live allocators for one stream, each unaware of the
|
|
1145
|
+
other, which is exactly the M6 invariant `close()` must not itself reopen."""
|
|
1146
|
+
with type(self)._registry_lock:
|
|
1147
|
+
key = (getattr(self, "root", None), getattr(self, "_name", None))
|
|
1148
|
+
if type(self)._registry.get(key) is self:
|
|
1149
|
+
del type(self)._registry[key]
|
|
1150
|
+
fd = getattr(self, "_lock_fd", None)
|
|
1151
|
+
if fd is not None:
|
|
1152
|
+
try:
|
|
1153
|
+
os.close(fd)
|
|
1154
|
+
except OSError:
|
|
1155
|
+
pass
|
|
1156
|
+
self._lock_fd = None
|
|
1157
|
+
self._closed = True
|
|
1158
|
+
|
|
1159
|
+
def _require_writable(self) -> None:
|
|
1160
|
+
"""P2/P7: the single gate `append`, `advance_watermark`, and `_rewrite_retaining` all call
|
|
1161
|
+
before touching disk -- refuses a closed, forked, or poisoned instance instead of letting it
|
|
1162
|
+
silently duplicate allocators, advance a parent's watermark from a forked child, or compound
|
|
1163
|
+
a rollback failure `_truncate_to` already flagged as unrecoverable."""
|
|
1164
|
+
if self._closed:
|
|
1165
|
+
raise OutboundSpoolError(
|
|
1166
|
+
"outbound_spool_closed",
|
|
1167
|
+
f"{self._name}: this OutboundSpool was closed; construct a new one for this stream",
|
|
1168
|
+
)
|
|
1169
|
+
if self._forked:
|
|
1170
|
+
raise OutboundSpoolError(
|
|
1171
|
+
"outbound_spool_forked",
|
|
1172
|
+
f"{self._name}: this OutboundSpool was constructed before a fork; construct a "
|
|
1173
|
+
f"new one in the child process instead of reusing this one",
|
|
1174
|
+
)
|
|
1175
|
+
if self._poisoned:
|
|
1176
|
+
raise OutboundSpoolError(
|
|
1177
|
+
"outbound_spool_poisoned",
|
|
1178
|
+
f"{self._name}: a prior rollback failed and left this spool in an unknown "
|
|
1179
|
+
f"state; it must not be mutated again",
|
|
1180
|
+
)
|
|
1181
|
+
|
|
1182
|
+
def _acquire_process_lock(self) -> int:
|
|
1183
|
+
"""M6: cross-PROCESS protection (the in-process registry above only protects against a
|
|
1184
|
+
second Python-level instance in this same interpreter). Held for the life of this instance
|
|
1185
|
+
-- released implicitly when its fd closes (process exit, `close()`, or `_forget_for_tests`
|
|
1186
|
+
in tests)."""
|
|
1187
|
+
if fcntl is None: # N30
|
|
1188
|
+
raise OutboundSpoolError(
|
|
1189
|
+
"outbound_spool_platform_unsupported",
|
|
1190
|
+
f"{self._name}: fcntl (POSIX file locking) is unavailable on this platform",
|
|
1191
|
+
)
|
|
1192
|
+
lock_path = self.root / f"{self._name}.spool.lock"
|
|
1193
|
+
fd = os.open(str(lock_path), os.O_CREAT | os.O_RDWR, 0o600)
|
|
1194
|
+
try:
|
|
1195
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
1196
|
+
except OSError as exc:
|
|
1197
|
+
os.close(fd)
|
|
1198
|
+
raise OutboundSpoolError(
|
|
1199
|
+
"outbound_spool_locked",
|
|
1200
|
+
f"{self._name}: already locked by another process",
|
|
1201
|
+
) from exc
|
|
1202
|
+
return fd
|
|
1203
|
+
|
|
1204
|
+
def _fsync_dir(self) -> None:
|
|
1205
|
+
fd = os.open(str(self.root), os.O_DIRECTORY)
|
|
1206
|
+
try:
|
|
1207
|
+
os.fsync(fd)
|
|
1208
|
+
finally:
|
|
1209
|
+
os.close(fd)
|
|
1210
|
+
|
|
1211
|
+
def _report_corruption(self, offset: int) -> None:
|
|
1212
|
+
"""N8: the client-visible half of "degrade, don't wedge" -- `is_corrupt`/`corruption_offset`
|
|
1213
|
+
stay true/set for the life of this instance once discovered, but the loud `logger.error`
|
|
1214
|
+
fires only the FIRST time (repeated reads of an already-known-corrupt spool would otherwise
|
|
1215
|
+
spam the log every flush cycle)."""
|
|
1216
|
+
with self._lock:
|
|
1217
|
+
already_reported = self._corrupt_since_offset is not None
|
|
1218
|
+
self._corrupt_since_offset = offset
|
|
1219
|
+
if not already_reported:
|
|
1220
|
+
logger.error(
|
|
1221
|
+
"%s: spool corrupt at byte offset %d -- the file is left untouched; only records "
|
|
1222
|
+
"before that offset are trusted. Reading degrades to the readable prefix rather "
|
|
1223
|
+
"than raising -- this is reported once per process, not per read.",
|
|
1224
|
+
self._name,
|
|
1225
|
+
offset,
|
|
1226
|
+
)
|
|
1227
|
+
|
|
1228
|
+
@property
|
|
1229
|
+
def is_corrupt(self) -> bool:
|
|
1230
|
+
"""N8: set once `_recover` or a later read finds a genuinely corrupt (not merely torn)
|
|
1231
|
+
record. A caller can turn this into a `log`-kind event; nothing in this class does that
|
|
1232
|
+
itself (no scenario/attempt context here to build one)."""
|
|
1233
|
+
return self._corrupt_since_offset is not None
|
|
1234
|
+
|
|
1235
|
+
@property
|
|
1236
|
+
def corruption_offset(self) -> int | None:
|
|
1237
|
+
return self._corrupt_since_offset
|
|
1238
|
+
|
|
1239
|
+
def _recover(self) -> None:
|
|
1240
|
+
# NOTE: the sequenced branch below must run even when the log is missing or empty -- an
|
|
1241
|
+
# M2-style lost file (durable watermark, vanished log) is exactly the case M5 needs to
|
|
1242
|
+
# reconcile against; an early return here would skip that reconciliation entirely and
|
|
1243
|
+
# silently reset the allocator to 1.
|
|
1244
|
+
records: list[tuple[int, bytes, Any]] = []
|
|
1245
|
+
if self._path.exists():
|
|
1246
|
+
data = self._path.read_bytes()
|
|
1247
|
+
if data:
|
|
1248
|
+
records, valid_length, corruption_offset = _iter_complete_records(data)
|
|
1249
|
+
if corruption_offset is not None:
|
|
1250
|
+
# N8: never truncate mid-file damage -- leave the bytes exactly as they are,
|
|
1251
|
+
# trust only what came before, and let construction succeed anyway (B1 already
|
|
1252
|
+
# forbids renumbering past it; the OTHER extreme -- raising here -- would
|
|
1253
|
+
# discard every future emit, including the terminal event, forever).
|
|
1254
|
+
self._report_corruption(corruption_offset)
|
|
1255
|
+
elif valid_length < len(data):
|
|
1256
|
+
with self._path.open("r+b") as stream:
|
|
1257
|
+
stream.truncate(valid_length)
|
|
1258
|
+
stream.flush()
|
|
1259
|
+
os.fsync(
|
|
1260
|
+
stream.fileno()
|
|
1261
|
+
) # MIN-11: durable, not left as a crash window
|
|
1262
|
+
if self._sequenced:
|
|
1263
|
+
max_sequence = 0
|
|
1264
|
+
offsets: dict[int, int] = {}
|
|
1265
|
+
for offset, _line, record in records:
|
|
1266
|
+
if isinstance(record, dict):
|
|
1267
|
+
sequence = record.get("sequence")
|
|
1268
|
+
if isinstance(sequence, int):
|
|
1269
|
+
offsets[sequence] = offset
|
|
1270
|
+
if sequence > max_sequence:
|
|
1271
|
+
max_sequence = sequence
|
|
1272
|
+
self._offset_by_sequence = offsets
|
|
1273
|
+
watermark = self.watermark()
|
|
1274
|
+
if watermark > max_sequence:
|
|
1275
|
+
logger.warning(
|
|
1276
|
+
"%s: watermark (%s) is ahead of the highest sequence found in the spool (%s) "
|
|
1277
|
+
"-- the log lost records the platform already processed; seeding "
|
|
1278
|
+
"next_sequence from the watermark so newly allocated sequences don't collide "
|
|
1279
|
+
"with ones the platform already closed",
|
|
1280
|
+
self._name,
|
|
1281
|
+
watermark,
|
|
1282
|
+
max_sequence,
|
|
1283
|
+
)
|
|
1284
|
+
self._next_sequence = max(max_sequence, watermark) + 1
|
|
1285
|
+
|
|
1286
|
+
def _truncate_to(self, size: int) -> None:
|
|
1287
|
+
"""B1: a true no-op on a failed append. `_next_sequence` is only advanced AFTER a
|
|
1288
|
+
successful write, so the retry reuses the same sequence number -- this makes sure it reuses
|
|
1289
|
+
clean ground too, instead of appending immediately after torn bytes with no `\\n` between
|
|
1290
|
+
them (which would merge into one unparseable line the rest of this class can't recover
|
|
1291
|
+
from)."""
|
|
1292
|
+
try:
|
|
1293
|
+
with self._path.open("r+b") as stream:
|
|
1294
|
+
stream.truncate(size)
|
|
1295
|
+
stream.flush()
|
|
1296
|
+
os.fsync(stream.fileno())
|
|
1297
|
+
except OSError as exc:
|
|
1298
|
+
# N12: the write may have failed before the file even existed (nothing to truncate,
|
|
1299
|
+
# harmless) OR the rollback itself failed on an existing torn write -- in the latter
|
|
1300
|
+
# case B1's "a failed append is a true no-op" no longer holds, so poison this spool
|
|
1301
|
+
# rather than let a future append silently merge into the torn bytes.
|
|
1302
|
+
self._poisoned = True
|
|
1303
|
+
logger.error(
|
|
1304
|
+
"%s: rollback of a failed append could not truncate the spool back to %d bytes "
|
|
1305
|
+
"(%s) -- the file may now carry torn bytes; poisoning this spool so a caller sees "
|
|
1306
|
+
"a typed error instead of a future append compounding the corruption",
|
|
1307
|
+
self._name,
|
|
1308
|
+
size,
|
|
1309
|
+
exc,
|
|
1310
|
+
)
|
|
1311
|
+
|
|
1312
|
+
def append(self, record: dict[str, Any]) -> SpooledRecord:
|
|
1313
|
+
with self._lock:
|
|
1314
|
+
self._require_writable()
|
|
1315
|
+
if self._sequenced:
|
|
1316
|
+
assigned = self._next_sequence
|
|
1317
|
+
record = {**record, "sequence": assigned}
|
|
1318
|
+
elif "sequence" in record:
|
|
1319
|
+
raise OutboundSpoolError(
|
|
1320
|
+
"outbound_spool_caller_supplied_sequence",
|
|
1321
|
+
f"{self._name} spool does not assign sequence numbers; caller must not pass one",
|
|
1322
|
+
)
|
|
1323
|
+
body = canonical_bytes(record)
|
|
1324
|
+
if b"\n" in body:
|
|
1325
|
+
# Framing invariant `_iter_complete_records`/`_recover` depend on: canonical JSON
|
|
1326
|
+
# never contains a raw newline (json.dumps escapes control characters inside
|
|
1327
|
+
# strings), so this would only fire on a value this module's own canonicalization
|
|
1328
|
+
# contract disallows.
|
|
1329
|
+
raise OutboundSpoolError(
|
|
1330
|
+
"outbound_spool_record_unframable",
|
|
1331
|
+
f"{self._name}: record contains a raw newline",
|
|
1332
|
+
)
|
|
1333
|
+
existed_before = self._path.exists()
|
|
1334
|
+
size_before = self._path.stat().st_size if existed_before else 0
|
|
1335
|
+
try:
|
|
1336
|
+
with self._path.open("ab") as stream:
|
|
1337
|
+
stream.write(body)
|
|
1338
|
+
stream.write(b"\n")
|
|
1339
|
+
stream.flush()
|
|
1340
|
+
os.fsync(stream.fileno())
|
|
1341
|
+
except BaseException:
|
|
1342
|
+
self._truncate_to(size_before)
|
|
1343
|
+
raise
|
|
1344
|
+
if not existed_before and not self._dir_synced:
|
|
1345
|
+
# M2: the directory entry for a brand-new file isn't durable just because the
|
|
1346
|
+
# file's own data is -- fsync it once (not per append; existing files' entries were
|
|
1347
|
+
# already synced by whichever append first created them).
|
|
1348
|
+
self._fsync_dir()
|
|
1349
|
+
self._dir_synced = True
|
|
1350
|
+
if self._sequenced:
|
|
1351
|
+
self._offset_by_sequence[assigned] = size_before
|
|
1352
|
+
self._next_sequence = assigned + 1
|
|
1353
|
+
return SpooledRecord(sequence=assigned, body=body)
|
|
1354
|
+
return SpooledRecord(sequence=None, body=body)
|
|
1355
|
+
|
|
1356
|
+
def records(self) -> list[SpooledRecord]:
|
|
1357
|
+
"""Reads the WHOLE log. The read itself happens OUTSIDE `self._lock` (M4): only the size
|
|
1358
|
+
snapshot that bounds it is taken under the lock, so the flusher's (potentially large) read
|
|
1359
|
+
never blocks `append`, the call loop's write path, for its duration. Safe because `append`
|
|
1360
|
+
only ever grows the file -- a read bounded to a size captured a moment earlier can only be
|
|
1361
|
+
stale, never torn. (`compact_through`/`drop` DO shrink the file and take the lock for their
|
|
1362
|
+
entire duration; this module assumes the flusher serializes its own reads against its own
|
|
1363
|
+
compactions rather than running them from two different threads.)
|
|
1364
|
+
"""
|
|
1365
|
+
with self._lock:
|
|
1366
|
+
if not self._path.exists():
|
|
1367
|
+
return []
|
|
1368
|
+
size = self._path.stat().st_size
|
|
1369
|
+
with self._path.open("rb") as stream:
|
|
1370
|
+
data = stream.read(size)
|
|
1371
|
+
parsed, _valid_length, corruption_offset = _iter_complete_records(data)
|
|
1372
|
+
if (
|
|
1373
|
+
corruption_offset is not None
|
|
1374
|
+
): # N8: degrade to the readable prefix, never raise here
|
|
1375
|
+
self._report_corruption(corruption_offset)
|
|
1376
|
+
out: list[SpooledRecord] = []
|
|
1377
|
+
for _offset, line, decoded in parsed:
|
|
1378
|
+
sequence = (
|
|
1379
|
+
decoded.get("sequence")
|
|
1380
|
+
if self._sequenced and isinstance(decoded, dict)
|
|
1381
|
+
else None
|
|
1382
|
+
)
|
|
1383
|
+
out.append(SpooledRecord(sequence=sequence, body=line))
|
|
1384
|
+
return out
|
|
1385
|
+
|
|
1386
|
+
def records_after(self, sequence: int) -> list[SpooledRecord]:
|
|
1387
|
+
if not self._sequenced:
|
|
1388
|
+
raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
|
|
1389
|
+
return [item for item in self.records() if (item.sequence or 0) > sequence]
|
|
1390
|
+
|
|
1391
|
+
@property
|
|
1392
|
+
def next_sequence(self) -> int:
|
|
1393
|
+
if not self._sequenced:
|
|
1394
|
+
raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
|
|
1395
|
+
assert self._next_sequence is not None
|
|
1396
|
+
return self._next_sequence
|
|
1397
|
+
|
|
1398
|
+
def watermark(self) -> int:
|
|
1399
|
+
"""The highest-processed sequence acknowledged so far ("the watermark is highest-processed
|
|
1400
|
+
-- accepted AND rejected sequences both advance it"). Durable across a restart via a
|
|
1401
|
+
fsync'd-temp-then-rename-then-fsync'd-directory write (M1) -- a corrupt or unreadable
|
|
1402
|
+
watermark file DEGRADES TO 0 with a loud diagnostic rather than raising and wedging the
|
|
1403
|
+
spool: re-sending already-acked events is safe (at-least-once delivery + platform-side
|
|
1404
|
+
dedupe on `event_id`), while a permanently unreadable outbound channel is not.
|
|
1405
|
+
"""
|
|
1406
|
+
if not self._sequenced:
|
|
1407
|
+
raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
|
|
1408
|
+
if not self._watermark_path.exists():
|
|
1409
|
+
return 0
|
|
1410
|
+
try:
|
|
1411
|
+
raw = json.loads(self._watermark_path.read_text(encoding="utf-8"))
|
|
1412
|
+
return int(raw["acked_through_sequence"])
|
|
1413
|
+
except (OSError, ValueError, KeyError, TypeError) as exc:
|
|
1414
|
+
logger.warning(
|
|
1415
|
+
"%s: watermark file is corrupt or unreadable (%s) -- degrading to 0. Re-sending "
|
|
1416
|
+
"already-acked events is safe (at-least-once + dedupe on event_id); wedging the "
|
|
1417
|
+
"spool permanently is not.",
|
|
1418
|
+
self._name,
|
|
1419
|
+
exc,
|
|
1420
|
+
)
|
|
1421
|
+
return 0
|
|
1422
|
+
|
|
1423
|
+
def advance_watermark(self, sequence: int) -> None:
|
|
1424
|
+
"""v1.3: `acked_through_sequence` is untrusted platform input. A value outside
|
|
1425
|
+
`[current_watermark, next_sequence)` is rejected locally with a typed error and the
|
|
1426
|
+
watermark is left exactly as it was -- "a malformed ack must not be able to discard
|
|
1427
|
+
pending records" (M7). `sequence == current_watermark` is a legitimate no-op, not an error
|
|
1428
|
+
(repeating the same ack, or a same-valued out-of-order response).
|
|
1429
|
+
|
|
1430
|
+
P7: this is the one operation that DESTROYS delivery state (it durably advances what a
|
|
1431
|
+
future `pending_since_watermark()` will ever return again) -- a forked child or a poisoned
|
|
1432
|
+
instance advancing it would silently orphan every pending record below the new value, the
|
|
1433
|
+
N1 outcome by a different route. `_require_writable()` guards it for that reason even
|
|
1434
|
+
though nothing here writes to the JSONL log itself.
|
|
1435
|
+
"""
|
|
1436
|
+
if not self._sequenced:
|
|
1437
|
+
raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
|
|
1438
|
+
with self._lock:
|
|
1439
|
+
self._require_writable()
|
|
1440
|
+
current = self.watermark()
|
|
1441
|
+
if sequence < current or sequence >= self._next_sequence:
|
|
1442
|
+
raise OutboundSpoolError(
|
|
1443
|
+
"outbound_spool_watermark_out_of_range",
|
|
1444
|
+
f"{self._name}: acked_through_sequence={sequence} outside the trusted range "
|
|
1445
|
+
f"[{current}, {self._next_sequence}) -- untrusted platform input, watermark "
|
|
1446
|
+
f"left unchanged",
|
|
1447
|
+
)
|
|
1448
|
+
if sequence == current:
|
|
1449
|
+
return
|
|
1450
|
+
temporary = (
|
|
1451
|
+
self.root
|
|
1452
|
+
/ f"{self._name}.spool.watermark.tmp.{os.getpid()}.{uuid.uuid4().hex}"
|
|
1453
|
+
)
|
|
1454
|
+
with temporary.open("wb") as stream:
|
|
1455
|
+
stream.write(
|
|
1456
|
+
json.dumps(
|
|
1457
|
+
{"acked_through_sequence": sequence}, separators=(",", ":")
|
|
1458
|
+
).encode("utf-8")
|
|
1459
|
+
)
|
|
1460
|
+
stream.flush()
|
|
1461
|
+
os.fsync(stream.fileno())
|
|
1462
|
+
os.replace(temporary, self._watermark_path)
|
|
1463
|
+
self._fsync_dir()
|
|
1464
|
+
|
|
1465
|
+
def pending_since_watermark(self) -> list[SpooledRecord]:
|
|
1466
|
+
"""Convenience for "the guest advances its spool cursor through the watermark": every
|
|
1467
|
+
spooled record not yet acknowledged, in sequence order. Uses the offset `append` recorded
|
|
1468
|
+
for the first pending sequence to SEEK directly there (M4) instead of re-reading and
|
|
1469
|
+
re-parsing the whole log on every flush cycle; falls back to the generic full scan when no
|
|
1470
|
+
cached offset exists yet (e.g. a fresh recovery whose watermark sits past every record this
|
|
1471
|
+
process itself has written an offset for).
|
|
1472
|
+
|
|
1473
|
+
N13: the drained steady state (nothing pending -- what a polling flusher sees most cycles)
|
|
1474
|
+
is checked first and returns `[]` with NO file IO at all: `watermark + 1 == next_sequence`
|
|
1475
|
+
means every allocated sequence has already been acknowledged, so there is nothing on disk
|
|
1476
|
+
to seek to regardless of what `_offset_by_sequence` does or doesn't have cached.
|
|
1477
|
+
"""
|
|
1478
|
+
watermark = self.watermark()
|
|
1479
|
+
with self._lock:
|
|
1480
|
+
if watermark + 1 == self._next_sequence:
|
|
1481
|
+
return []
|
|
1482
|
+
offset = self._offset_by_sequence.get(watermark + 1)
|
|
1483
|
+
if offset is None:
|
|
1484
|
+
return self.records_after(watermark)
|
|
1485
|
+
with self._lock:
|
|
1486
|
+
if not self._path.exists():
|
|
1487
|
+
return []
|
|
1488
|
+
size = self._path.stat().st_size
|
|
1489
|
+
if offset >= size:
|
|
1490
|
+
return []
|
|
1491
|
+
with self._path.open("rb") as stream:
|
|
1492
|
+
stream.seek(offset)
|
|
1493
|
+
data = stream.read(size - offset)
|
|
1494
|
+
parsed, _valid_length, corruption_offset = _iter_complete_records(data)
|
|
1495
|
+
if (
|
|
1496
|
+
corruption_offset is not None
|
|
1497
|
+
): # N8: absolute offset -- `data` starts at `offset`
|
|
1498
|
+
self._report_corruption(offset + corruption_offset)
|
|
1499
|
+
return [
|
|
1500
|
+
SpooledRecord(
|
|
1501
|
+
sequence=decoded.get("sequence") if isinstance(decoded, dict) else None,
|
|
1502
|
+
body=line,
|
|
1503
|
+
)
|
|
1504
|
+
for _offset, line, decoded in parsed
|
|
1505
|
+
]
|
|
1506
|
+
|
|
1507
|
+
def _rewrite_retaining(self, keep: Callable[[Any], bool]) -> None:
|
|
1508
|
+
"""Shared by `compact_through` and `drop_many`: rewrites the log keeping only records
|
|
1509
|
+
`keep` accepts, via the same write-fsync/atomic-replace/fsync-directory durability shape
|
|
1510
|
+
`append`/`advance_watermark` use -- a crash mid-rewrite leaves either the old file intact
|
|
1511
|
+
or the new one complete, never a torn hybrid.
|
|
1512
|
+
|
|
1513
|
+
P1: if the log is already corrupt (N8), this REFUSES instead of rewriting. A rewrite always
|
|
1514
|
+
replaces the file from what it read, and reading stops at the corruption offset -- so
|
|
1515
|
+
rewriting on a corrupt spool would not merely skip the corrupt bytes, it would silently
|
|
1516
|
+
destroy every intact record PAST them too, including ones this process itself appended
|
|
1517
|
+
after recovery that have never been sent (the terminal event, in the worst case). That
|
|
1518
|
+
directly defeats N8's "the readable prefix stays usable, delivery keeps working" posture
|
|
1519
|
+
the moment the first drop/compact happens. Refusing costs only unbounded disk growth until
|
|
1520
|
+
the attempt ends (`compact_through`'s whole job) or a rejected record staying spooled but
|
|
1521
|
+
never re-emitted anyway, since it's at or below the watermark (`drop_many`'s whole job) --
|
|
1522
|
+
both strictly better than deleting undelivered records.
|
|
1523
|
+
"""
|
|
1524
|
+
with self._lock:
|
|
1525
|
+
self._require_writable()
|
|
1526
|
+
if not self._path.exists():
|
|
1527
|
+
return
|
|
1528
|
+
size = self._path.stat().st_size
|
|
1529
|
+
with self._path.open("rb") as stream:
|
|
1530
|
+
data = stream.read(size)
|
|
1531
|
+
parsed, _valid_length, corruption_offset = _iter_complete_records(data)
|
|
1532
|
+
if corruption_offset is not None:
|
|
1533
|
+
self._report_corruption(corruption_offset)
|
|
1534
|
+
logger.error(
|
|
1535
|
+
"%s: refusing to compact/drop on a corrupt spool -- a rewrite replaces the file "
|
|
1536
|
+
"from what it read, and reading stops at byte %d, so every record past that "
|
|
1537
|
+
"offset (including not-yet-delivered ones) would be destroyed. The log is left "
|
|
1538
|
+
"intact and grows unbounded until the attempt ends; that is the fail-safe half "
|
|
1539
|
+
"of degrade-not-wedge.",
|
|
1540
|
+
self._name,
|
|
1541
|
+
corruption_offset,
|
|
1542
|
+
)
|
|
1543
|
+
return
|
|
1544
|
+
temporary = (
|
|
1545
|
+
self.root
|
|
1546
|
+
/ f"{self._name}.spool.jsonl.tmp.{os.getpid()}.{uuid.uuid4().hex}"
|
|
1547
|
+
)
|
|
1548
|
+
offsets: dict[int, int] = {}
|
|
1549
|
+
offset = 0
|
|
1550
|
+
with temporary.open("wb") as stream:
|
|
1551
|
+
for _old_offset, line, decoded in parsed:
|
|
1552
|
+
if not keep(decoded):
|
|
1553
|
+
continue
|
|
1554
|
+
if (
|
|
1555
|
+
self._sequenced
|
|
1556
|
+
and isinstance(decoded, dict)
|
|
1557
|
+
and isinstance(decoded.get("sequence"), int)
|
|
1558
|
+
):
|
|
1559
|
+
offsets[decoded["sequence"]] = offset
|
|
1560
|
+
stream.write(line)
|
|
1561
|
+
stream.write(b"\n")
|
|
1562
|
+
offset += len(line) + 1
|
|
1563
|
+
stream.flush()
|
|
1564
|
+
os.fsync(stream.fileno())
|
|
1565
|
+
os.replace(temporary, self._path)
|
|
1566
|
+
self._fsync_dir()
|
|
1567
|
+
if self._sequenced:
|
|
1568
|
+
self._offset_by_sequence = offsets
|
|
1569
|
+
|
|
1570
|
+
def compact_through(self, sequence: int) -> None:
|
|
1571
|
+
"""M4: physically drops every durably-acked record (`sequence <= min(sequence,
|
|
1572
|
+
watermark())`) from the on-disk log, bounding its growth for a long `running` stage. The
|
|
1573
|
+
allocator's `next_sequence` is unaffected -- it is only ever derived from the log at
|
|
1574
|
+
`_recover` time, and recovery's own `max(max_sequence, watermark)` rule (M5) already
|
|
1575
|
+
tolerates a log whose historical records were compacted away, since none of them can be
|
|
1576
|
+
the true maximum (compaction only ever removes sequences at or below the watermark, and
|
|
1577
|
+
the watermark is always <= every pending, uncompacted sequence).
|
|
1578
|
+
|
|
1579
|
+
Clamped to the current watermark regardless of what the caller passes -- compacting past
|
|
1580
|
+
an event the platform hasn't actually processed yet would be irreversible data loss, and
|
|
1581
|
+
this module's posture throughout is fail-safe over trusting the caller.
|
|
1582
|
+
"""
|
|
1583
|
+
if not self._sequenced:
|
|
1584
|
+
raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
|
|
1585
|
+
effective = min(sequence, self.watermark())
|
|
1586
|
+
self._rewrite_retaining(
|
|
1587
|
+
lambda decoded: (
|
|
1588
|
+
not (
|
|
1589
|
+
isinstance(decoded, dict)
|
|
1590
|
+
and isinstance(decoded.get("sequence"), int)
|
|
1591
|
+
and decoded["sequence"] <= effective
|
|
1592
|
+
)
|
|
1593
|
+
)
|
|
1594
|
+
)
|
|
1595
|
+
|
|
1596
|
+
def drop_many(self, sequences: Collection[int]) -> None:
|
|
1597
|
+
"""The contract's rejected-event mechanism: "a rejected event is dropped from the spool ...
|
|
1598
|
+
it is never re-emitted" (M5). PURE physical removal (N1) -- every sequence in `sequences`
|
|
1599
|
+
is deleted from the on-disk log in ONE rewrite pass (N14: a batch of 100 rejections is one
|
|
1600
|
+
`_rewrite_retaining` call, not 100), and the watermark is left untouched.
|
|
1601
|
+
|
|
1602
|
+
N1: an earlier version had `drop` also advance the watermark to `sequence`, on the theory
|
|
1603
|
+
that "a rejected event closes its sequence." That let an UNTRUSTED `rejected[].sequence`
|
|
1604
|
+
from the platform silently orphan every pending record below it, bypassing
|
|
1605
|
+
`advance_watermark`'s own M7 clamp entirely -- the clamp only guards `acked_through_sequence`
|
|
1606
|
+
callers, and `drop` skipped straight past it. The batch-level
|
|
1607
|
+
`advance_watermark(acked_through_sequence)` a caller performs separately is the ONE place
|
|
1608
|
+
the watermark ever moves; it already covers every rejected sequence under a conformant
|
|
1609
|
+
platform (rejections advance the watermark by contract), and under a non-conformant one
|
|
1610
|
+
M7's clamp is then the single, correct chokepoint -- this method has no clamp of its own to
|
|
1611
|
+
bypass.
|
|
1612
|
+
|
|
1613
|
+
Writing the record's payload to the artifact spool as a `log` kind (the other half of the
|
|
1614
|
+
contract's drop rule) is NOT this method's job -- that hand-off needs scenario/attempt
|
|
1615
|
+
context and an `ArtifactsClient` this layer doesn't own; it is P9/P10 wiring, documented
|
|
1616
|
+
here as the seam rather than guessed at.
|
|
1617
|
+
"""
|
|
1618
|
+
if not self._sequenced:
|
|
1619
|
+
raise OutboundSpoolError("outbound_spool_unsequenced", self._name)
|
|
1620
|
+
sequence_set = set(sequences)
|
|
1621
|
+
if not sequence_set:
|
|
1622
|
+
return
|
|
1623
|
+
self._rewrite_retaining(
|
|
1624
|
+
lambda decoded: (
|
|
1625
|
+
not (
|
|
1626
|
+
isinstance(decoded, dict)
|
|
1627
|
+
and decoded.get("sequence") in sequence_set
|
|
1628
|
+
)
|
|
1629
|
+
)
|
|
1630
|
+
)
|
|
1631
|
+
|
|
1632
|
+
def drop(self, sequence: int) -> None:
|
|
1633
|
+
"""Single-sequence convenience wrapper over `drop_many` -- see its docstring for why this
|
|
1634
|
+
no longer touches the watermark."""
|
|
1635
|
+
self.drop_many((sequence,))
|
|
1636
|
+
|
|
1637
|
+
|
|
1638
|
+
def _poison_after_fork() -> None:
|
|
1639
|
+
"""N25: `os.fork()` inherits both `OutboundSpool._registry` (with a live `_next_sequence`) and
|
|
1640
|
+
every instance's flock fd (the SAME open file description, so the lock is merely shared, not
|
|
1641
|
+
contended, across parent and child) -- without this, parent and child would allocate identical
|
|
1642
|
+
sequence numbers with no complaint. Marks every currently-registered instance so its next
|
|
1643
|
+
mutating call raises instead. Not reachable via `subprocess` (fork+exec resets memory); this
|
|
1644
|
+
guards a bare `os.fork()` specifically."""
|
|
1645
|
+
with OutboundSpool._registry_lock:
|
|
1646
|
+
for instance in OutboundSpool._registry.values():
|
|
1647
|
+
instance._forked = True
|
|
1648
|
+
|
|
1649
|
+
|
|
1650
|
+
if hasattr(os, "register_at_fork"): # POSIX-only, like fcntl (N30)
|
|
1651
|
+
# P7: `before=`/`after_in_parent=` pair the registry lock around the fork itself -- without
|
|
1652
|
+
# this, a fork occurring while some OTHER thread holds `_registry_lock` hands the child a
|
|
1653
|
+
# locked RLock owned by a thread that no longer exists there, and `_poison_after_fork`'s own
|
|
1654
|
+
# `with OutboundSpool._registry_lock:` deadlocks at the fork point instead of poisoning
|
|
1655
|
+
# anything. Acquiring on `before` guarantees the FORKING thread itself owns the lock at fork
|
|
1656
|
+
# time, so the child's single surviving thread already owns it too -- `_poison_after_fork`'s
|
|
1657
|
+
# acquire becomes a safe reentrant no-op there, and `after_in_parent` restores normal locking
|
|
1658
|
+
# in the parent.
|
|
1659
|
+
os.register_at_fork(
|
|
1660
|
+
before=OutboundSpool._registry_lock.acquire,
|
|
1661
|
+
after_in_parent=OutboundSpool._registry_lock.release,
|
|
1662
|
+
after_in_child=_poison_after_fork,
|
|
1663
|
+
)
|
|
1664
|
+
|
|
1665
|
+
|
|
1666
|
+
# =================================================================================================
|
|
1667
|
+
# Transport -- the HTTP boundary every channel client speaks through, and its production impl.
|
|
1668
|
+
# =================================================================================================
|
|
1669
|
+
|
|
1670
|
+
|
|
1671
|
+
@dataclass(frozen=True)
|
|
1672
|
+
class TransportResponse:
|
|
1673
|
+
status_code: int
|
|
1674
|
+
body: dict[str, Any] | None
|
|
1675
|
+
headers: dict[str, str]
|
|
1676
|
+
|
|
1677
|
+
|
|
1678
|
+
class TransportError(RuntimeError):
|
|
1679
|
+
"""Raised by a `Transport.request` implementation when no HTTP response was ever received
|
|
1680
|
+
(connection refused, DNS failure, timeout, ...). `classify_response` treats this identically
|
|
1681
|
+
to an unreachable 5xx -- the guest cannot distinguish "server errored" from "server unreachable"
|
|
1682
|
+
and the contract's retry policy doesn't ask it to."""
|
|
1683
|
+
|
|
1684
|
+
|
|
1685
|
+
class Transport(Protocol):
|
|
1686
|
+
"""The seam every channel client is built against, so the fake-platform tests exercise the
|
|
1687
|
+
exact same code path production traffic does -- only what sits behind this protocol differs.
|
|
1688
|
+
"""
|
|
1689
|
+
|
|
1690
|
+
def request(
|
|
1691
|
+
self,
|
|
1692
|
+
method: str,
|
|
1693
|
+
url: str,
|
|
1694
|
+
*,
|
|
1695
|
+
headers: dict[str, str],
|
|
1696
|
+
json_body: dict[str, Any] | None = None,
|
|
1697
|
+
data: bytes | Iterator[bytes] | None = None,
|
|
1698
|
+
timeout: float = 30.0,
|
|
1699
|
+
) -> TransportResponse: ...
|
|
1700
|
+
|
|
1701
|
+
|
|
1702
|
+
class RequestsTransport:
|
|
1703
|
+
"""Production `Transport`: a thin `requests.Session` wrapper. Every network-layer failure
|
|
1704
|
+
(`requests.RequestException`, which covers connection errors, timeouts, and retries `requests`
|
|
1705
|
+
itself doesn't handle) is normalized to `TransportError` so `classify_response` never needs to
|
|
1706
|
+
know which HTTP library is underneath."""
|
|
1707
|
+
|
|
1708
|
+
def __init__(self, *, session: requests.Session | None = None) -> None:
|
|
1709
|
+
self._session = session or requests.Session()
|
|
1710
|
+
|
|
1711
|
+
def request(
|
|
1712
|
+
self,
|
|
1713
|
+
method: str,
|
|
1714
|
+
url: str,
|
|
1715
|
+
*,
|
|
1716
|
+
headers: dict[str, str],
|
|
1717
|
+
json_body: dict[str, Any] | None = None,
|
|
1718
|
+
data: bytes | Iterator[bytes] | None = None,
|
|
1719
|
+
timeout: float = 30.0,
|
|
1720
|
+
) -> TransportResponse:
|
|
1721
|
+
try:
|
|
1722
|
+
response = self._session.request(
|
|
1723
|
+
method, url, headers=headers, json=json_body, data=data, timeout=timeout
|
|
1724
|
+
)
|
|
1725
|
+
except requests.RequestException as exc:
|
|
1726
|
+
raise TransportError(str(exc)) from exc
|
|
1727
|
+
try:
|
|
1728
|
+
body = response.json() if response.content else None
|
|
1729
|
+
except ValueError:
|
|
1730
|
+
body = None
|
|
1731
|
+
return TransportResponse(
|
|
1732
|
+
status_code=response.status_code, body=body, headers=dict(response.headers)
|
|
1733
|
+
)
|
|
1734
|
+
|
|
1735
|
+
|
|
1736
|
+
def _iter_chunks(data: bytes, chunk_size: int) -> Iterator[bytes]:
|
|
1737
|
+
"""§3a: "Uploads >64 MB use chunked transfer." A fresh generator is built per send attempt
|
|
1738
|
+
(never reused across a retry) -- a generator is single-use, and reusing an exhausted one would
|
|
1739
|
+
silently upload an empty body on the second attempt."""
|
|
1740
|
+
for start in range(0, len(data), chunk_size):
|
|
1741
|
+
yield data[start : start + chunk_size]
|
|
1742
|
+
|
|
1743
|
+
|
|
1744
|
+
def _parse_retry_after(headers: Mapping[str, str] | None) -> float | None:
|
|
1745
|
+
""" "429 -> honor `Retry-After`." Only the delta-seconds form is parsed (the integer count of
|
|
1746
|
+
seconds to wait) -- the contract never mentions the alternative HTTP-date form and every
|
|
1747
|
+
platform emitter in this ecosystem is expected to send the simple form; an unparseable value is
|
|
1748
|
+
treated as absent so the caller falls back to the computed backoff rather than crashing.
|
|
1749
|
+
|
|
1750
|
+
N5: a negative value is ALSO treated as absent -- `time.sleep(-5)` raises `ValueError`, and a
|
|
1751
|
+
server sending a negative `Retry-After` is malformed input this module owes no obedience to.
|
|
1752
|
+
The upper clamp (`[0, retry_policy.max_backoff_seconds]`) needs the policy, which isn't
|
|
1753
|
+
available here -- `_perform_with_retry` applies that half.
|
|
1754
|
+
|
|
1755
|
+
P6: header-name lookup is case-INsensitive (RFC 9110 §5.1 -- field names are case-insensitive).
|
|
1756
|
+
`RequestsTransport` builds `dict(response.headers)` from `requests`' own `CaseInsensitiveDict`,
|
|
1757
|
+
which drops the case-insensitivity and preserves whatever casing the server actually sent -- a
|
|
1758
|
+
plain `.get("Retry-After")` would miss `RETRY-AFTER`/`Retry-after` and silently fall back to
|
|
1759
|
+
computed backoff instead of honoring the server's wait.
|
|
1760
|
+
"""
|
|
1761
|
+
if not headers:
|
|
1762
|
+
return None
|
|
1763
|
+
value = next((v for k, v in headers.items() if k.lower() == "retry-after"), None)
|
|
1764
|
+
if value is None:
|
|
1765
|
+
return None
|
|
1766
|
+
try:
|
|
1767
|
+
parsed = float(value)
|
|
1768
|
+
except ValueError:
|
|
1769
|
+
return None
|
|
1770
|
+
return parsed if parsed >= 0 else None
|
|
1771
|
+
|
|
1772
|
+
|
|
1773
|
+
# =================================================================================================
|
|
1774
|
+
# Error map -- the contract's closed status-code vocabulary ("Error responses" / "Failure semantics
|
|
1775
|
+
# summary"), and the shared retry engine every channel client drives it through.
|
|
1776
|
+
# =================================================================================================
|
|
1777
|
+
|
|
1778
|
+
|
|
1779
|
+
class ChannelOutcome(str, Enum):
|
|
1780
|
+
"""Every way one outbound HTTP attempt can resolve, per the contract's failure table. Not a
|
|
1781
|
+
contract vocabulary itself (the wire only ever carries a status code + `{error, message,
|
|
1782
|
+
retryable}`) -- this is this module's own closed classification of that table, the thing
|
|
1783
|
+
`classify_response` computes and every client branches on."""
|
|
1784
|
+
|
|
1785
|
+
DELIVERED = "delivered"
|
|
1786
|
+
RETRYABLE = "retryable"
|
|
1787
|
+
FENCED = "fenced"
|
|
1788
|
+
PERMANENT_ITEM = "permanent_item"
|
|
1789
|
+
CHANNEL_FAILED = "channel_failed"
|
|
1790
|
+
BUDGET_EXCEEDED = "budget_exceeded"
|
|
1791
|
+
|
|
1792
|
+
|
|
1793
|
+
@dataclass(frozen=True)
|
|
1794
|
+
class ChannelError:
|
|
1795
|
+
outcome: ChannelOutcome
|
|
1796
|
+
domain: FailureDomain | None
|
|
1797
|
+
code: str
|
|
1798
|
+
message: str
|
|
1799
|
+
retry_after_seconds: float | None = None
|
|
1800
|
+
# N7: the raw status this was classified from (`None` for a `TransportError`/no-response
|
|
1801
|
+
# outcome) -- carried so a caller can recognize a specific status (413, for the events-batch
|
|
1802
|
+
# halving retry) without `ChannelOutcome`/`code` alone being expressive enough for that.
|
|
1803
|
+
status_code: int | None = None
|
|
1804
|
+
|
|
1805
|
+
|
|
1806
|
+
class HostedFencedError(RuntimeError):
|
|
1807
|
+
"""401 (expired) / 403 (fence, scope, mismatch): "stop emitting, exit code 3 ... never an infra
|
|
1808
|
+
retry." Raised by the shared retry engine and never retried -- the entrypoint (outside this
|
|
1809
|
+
module) is the one that translates this into the process exit code."""
|
|
1810
|
+
|
|
1811
|
+
def __init__(self, error: ChannelError) -> None:
|
|
1812
|
+
self.error = error
|
|
1813
|
+
super().__init__(f"{error.code}: {error.message}")
|
|
1814
|
+
|
|
1815
|
+
|
|
1816
|
+
class HostedChannelFailedError(RuntimeError):
|
|
1817
|
+
"""404, retried 3x per the contract, still 404: "finalize `platform_sync`." Raised by the
|
|
1818
|
+
shared retry engine once `classify_response` reaches the third 404 attempt."""
|
|
1819
|
+
|
|
1820
|
+
def __init__(self, error: ChannelError) -> None:
|
|
1821
|
+
self.error = error
|
|
1822
|
+
super().__init__(f"{error.code}: {error.message}")
|
|
1823
|
+
|
|
1824
|
+
|
|
1825
|
+
class HostedAttemptSupersededError(RuntimeError):
|
|
1826
|
+
"""N22: `409 attempt_superseded` folded into the `ChannelState` latch below -- "a fenced
|
|
1827
|
+
attempt's in-flight requests cannot land after registration of its successor" is a fence in
|
|
1828
|
+
substance, even though the ONE request that received it is still correctly classified
|
|
1829
|
+
`PERMANENT_ITEM` (contract-correct: 409 is item-level, not fence-level). Only raised by
|
|
1830
|
+
`ChannelState.check()` on a LATER call, once a prior call has already seen this code -- the
|
|
1831
|
+
call that actually observed the 409 still returns its normal item-level result."""
|
|
1832
|
+
|
|
1833
|
+
def __init__(self, error: ChannelError) -> None:
|
|
1834
|
+
self.error = error
|
|
1835
|
+
super().__init__(f"{error.code}: {error.message}")
|
|
1836
|
+
|
|
1837
|
+
|
|
1838
|
+
class ChannelState:
|
|
1839
|
+
"""N10: shared "stop emitting" latch across the three channel clients for one attempt -- a
|
|
1840
|
+
fence (401/403) or an exhausted channel (404x3) on ANY channel must stop ALL of them, since
|
|
1841
|
+
the token/fence is per-ATTEMPT, not per-channel ("stop emitting ... never an infra retry").
|
|
1842
|
+
Also carries the N22 attempt-supersession latch (409 `attempt_superseded`).
|
|
1843
|
+
|
|
1844
|
+
Construct ONE `ChannelState` per attempt and pass it to `EventsClient`/`ResultsClient`/
|
|
1845
|
+
`ArtifactsClient` alike (each defaults to a private one if not given, which only latches
|
|
1846
|
+
itself -- correct for a single-channel caller, but callers driving more than one channel for
|
|
1847
|
+
the same attempt MUST share one instance to get the cross-channel guarantee this class exists
|
|
1848
|
+
for). Once latched, `check()` raises the SAME error on every subsequent call, from any client
|
|
1849
|
+
sharing this state, without ever touching the transport.
|
|
1850
|
+
"""
|
|
1851
|
+
|
|
1852
|
+
def __init__(self) -> None:
|
|
1853
|
+
self._error: (
|
|
1854
|
+
HostedFencedError
|
|
1855
|
+
| HostedChannelFailedError
|
|
1856
|
+
| HostedAttemptSupersededError
|
|
1857
|
+
| None
|
|
1858
|
+
) = None
|
|
1859
|
+
self._lock = RLock()
|
|
1860
|
+
|
|
1861
|
+
def check(self) -> None:
|
|
1862
|
+
with self._lock:
|
|
1863
|
+
error = self._error
|
|
1864
|
+
if error is not None:
|
|
1865
|
+
raise error
|
|
1866
|
+
|
|
1867
|
+
def latch(
|
|
1868
|
+
self,
|
|
1869
|
+
exc: "HostedFencedError | HostedChannelFailedError | HostedAttemptSupersededError",
|
|
1870
|
+
) -> None:
|
|
1871
|
+
with self._lock:
|
|
1872
|
+
if self._error is None:
|
|
1873
|
+
self._error = exc
|
|
1874
|
+
|
|
1875
|
+
|
|
1876
|
+
def classify_response(
|
|
1877
|
+
status_code: int | None,
|
|
1878
|
+
body: dict[str, Any] | None,
|
|
1879
|
+
*,
|
|
1880
|
+
attempt: int,
|
|
1881
|
+
retry_after_seconds: float | None = None,
|
|
1882
|
+
) -> ChannelError | None:
|
|
1883
|
+
"""The single call site every channel client classifies a transport outcome through. Returns
|
|
1884
|
+
`None` for a delivered (2xx) response, a `ChannelError` for everything else. `status_code=None`
|
|
1885
|
+
means `TransportError` (no response was ever received) -- classified exactly like an
|
|
1886
|
+
unreachable 5xx. For the 404 branch specifically, `attempt` must be the count of 404 RESPONSES
|
|
1887
|
+
seen so far in this call (not the overall attempt number) -- `_perform_with_retry` tracks that
|
|
1888
|
+
counter separately (N6) so an interleaved 5xx never shortens the 404 budget; every other branch
|
|
1889
|
+
ignores `attempt` entirely.
|
|
1890
|
+
|
|
1891
|
+
N28: the error body's `retryable` field is deliberately never read here -- classification is
|
|
1892
|
+
status-keyed throughout this module (every branch below is keyed on `status_code` alone), which
|
|
1893
|
+
is defensible per the contract's own closed, status-code-driven failure table; if the platform
|
|
1894
|
+
ever marks an unexpected status `retryable` the guest disagrees silently, a known, accepted gap.
|
|
1895
|
+
|
|
1896
|
+
Every branch below is transcribed directly from the contract's "Error responses" paragraph and
|
|
1897
|
+
"Failure semantics summary" table -- see `outbound-channels.md` v1.3 for the prose this mirrors.
|
|
1898
|
+
"""
|
|
1899
|
+
if status_code is not None and 200 <= status_code < 300:
|
|
1900
|
+
return None
|
|
1901
|
+
|
|
1902
|
+
error_code = (body or {}).get("error") if body else None
|
|
1903
|
+
message = (body or {}).get("message", "") if body else ""
|
|
1904
|
+
|
|
1905
|
+
if status_code is None:
|
|
1906
|
+
return ChannelError(
|
|
1907
|
+
ChannelOutcome.RETRYABLE,
|
|
1908
|
+
FailureDomain.CONNECTIVITY,
|
|
1909
|
+
error_code or "network_error",
|
|
1910
|
+
message or "transport failure: no response received",
|
|
1911
|
+
status_code=None,
|
|
1912
|
+
)
|
|
1913
|
+
if status_code in (401, 403):
|
|
1914
|
+
return ChannelError(
|
|
1915
|
+
ChannelOutcome.FENCED,
|
|
1916
|
+
None,
|
|
1917
|
+
error_code or "fenced",
|
|
1918
|
+
message,
|
|
1919
|
+
status_code=status_code,
|
|
1920
|
+
)
|
|
1921
|
+
if status_code == 404:
|
|
1922
|
+
# N29: PLATFORM_SYNC on every 404 attempt, not just the third -- a 404 is never a
|
|
1923
|
+
# connectivity fault under §4.6; only the third attempt's outcome is ever surfaced to a
|
|
1924
|
+
# caller, but the domain should not silently disagree across attempts 1-2 vs 3.
|
|
1925
|
+
domain = FailureDomain.PLATFORM_SYNC
|
|
1926
|
+
if attempt < 3:
|
|
1927
|
+
return ChannelError(
|
|
1928
|
+
ChannelOutcome.RETRYABLE,
|
|
1929
|
+
domain,
|
|
1930
|
+
error_code or "not_found",
|
|
1931
|
+
message,
|
|
1932
|
+
status_code=404,
|
|
1933
|
+
)
|
|
1934
|
+
return ChannelError(
|
|
1935
|
+
ChannelOutcome.CHANNEL_FAILED,
|
|
1936
|
+
domain,
|
|
1937
|
+
error_code or "not_found",
|
|
1938
|
+
message,
|
|
1939
|
+
status_code=404,
|
|
1940
|
+
)
|
|
1941
|
+
if status_code == 413:
|
|
1942
|
+
# N7: 413 is Channel 3's artifact-budget code specifically -- only classify it
|
|
1943
|
+
# BUDGET_EXCEEDED when the body actually says so; a 413 on any other channel (e.g. an
|
|
1944
|
+
# events batch that simply exceeded the platform's ingress size limit) falls through to
|
|
1945
|
+
# the unlisted-4xx catch-all below instead of being mislabeled as a budget condition that
|
|
1946
|
+
# channel doesn't have.
|
|
1947
|
+
if error_code == "artifact_budget_exceeded":
|
|
1948
|
+
return ChannelError(
|
|
1949
|
+
ChannelOutcome.BUDGET_EXCEEDED,
|
|
1950
|
+
None,
|
|
1951
|
+
error_code,
|
|
1952
|
+
message,
|
|
1953
|
+
status_code=413,
|
|
1954
|
+
)
|
|
1955
|
+
return ChannelError(
|
|
1956
|
+
ChannelOutcome.PERMANENT_ITEM,
|
|
1957
|
+
None,
|
|
1958
|
+
error_code or "http_413",
|
|
1959
|
+
message,
|
|
1960
|
+
status_code=413,
|
|
1961
|
+
)
|
|
1962
|
+
if status_code == 429:
|
|
1963
|
+
return ChannelError(
|
|
1964
|
+
ChannelOutcome.RETRYABLE,
|
|
1965
|
+
FailureDomain.CONNECTIVITY,
|
|
1966
|
+
error_code or "rate_limited",
|
|
1967
|
+
message,
|
|
1968
|
+
retry_after_seconds=retry_after_seconds,
|
|
1969
|
+
status_code=429,
|
|
1970
|
+
)
|
|
1971
|
+
if status_code in (400, 409, 422):
|
|
1972
|
+
return ChannelError(
|
|
1973
|
+
ChannelOutcome.PERMANENT_ITEM,
|
|
1974
|
+
None,
|
|
1975
|
+
error_code or f"http_{status_code}",
|
|
1976
|
+
message,
|
|
1977
|
+
status_code=status_code,
|
|
1978
|
+
)
|
|
1979
|
+
if 500 <= status_code < 600:
|
|
1980
|
+
return ChannelError(
|
|
1981
|
+
ChannelOutcome.RETRYABLE,
|
|
1982
|
+
FailureDomain.CONNECTIVITY,
|
|
1983
|
+
error_code or "server_error",
|
|
1984
|
+
message,
|
|
1985
|
+
status_code=status_code,
|
|
1986
|
+
)
|
|
1987
|
+
if 400 <= status_code < 500:
|
|
1988
|
+
# "Catch-all: any unlisted 4xx is permanent for that item."
|
|
1989
|
+
return ChannelError(
|
|
1990
|
+
ChannelOutcome.PERMANENT_ITEM,
|
|
1991
|
+
None,
|
|
1992
|
+
error_code or f"http_{status_code}",
|
|
1993
|
+
message,
|
|
1994
|
+
status_code=status_code,
|
|
1995
|
+
)
|
|
1996
|
+
# N27: no other status family is contractual -- notably a 3xx, which should never occur (every
|
|
1997
|
+
# endpoint ends in "/" precisely so Django's POST-redirect problem never arises). Treat as
|
|
1998
|
+
# permanent rather than retrying an endpoint misconfiguration `max_attempts` times before
|
|
1999
|
+
# giving up anyway.
|
|
2000
|
+
return ChannelError(
|
|
2001
|
+
ChannelOutcome.PERMANENT_ITEM,
|
|
2002
|
+
None,
|
|
2003
|
+
error_code or f"http_{status_code}",
|
|
2004
|
+
message,
|
|
2005
|
+
status_code=status_code,
|
|
2006
|
+
)
|
|
2007
|
+
|
|
2008
|
+
|
|
2009
|
+
def compute_backoff_seconds(
|
|
2010
|
+
attempt: int,
|
|
2011
|
+
*,
|
|
2012
|
+
initial_backoff_seconds: float,
|
|
2013
|
+
max_backoff_seconds: float,
|
|
2014
|
+
rng: Callable[[], float] = random.random,
|
|
2015
|
+
) -> float:
|
|
2016
|
+
""" "retry with backoff (base `retry.initial_backoff_seconds`, cap `retry.max_backoff_seconds`,
|
|
2017
|
+
full jitter)" -- `uniform(0, min(cap, base * 2**(attempt-1)))`. `attempt` is the 1-based attempt
|
|
2018
|
+
that just failed. `rng` is injectable so callers (and tests) can get a deterministic value.
|
|
2019
|
+
"""
|
|
2020
|
+
ceiling = min(
|
|
2021
|
+
max_backoff_seconds, initial_backoff_seconds * (2 ** max(0, attempt - 1))
|
|
2022
|
+
)
|
|
2023
|
+
return rng() * ceiling
|
|
2024
|
+
|
|
2025
|
+
|
|
2026
|
+
@dataclass(frozen=True)
|
|
2027
|
+
class RetryPolicy:
|
|
2028
|
+
"""Backoff parameters, shared by all three channel clients. Field names mirror
|
|
2029
|
+
`job.HarnessRetryPolicy` deliberately (the contract states these come from the job's own
|
|
2030
|
+
`retry.initial_backoff_seconds`/`retry.max_backoff_seconds`), but this module does not import
|
|
2031
|
+
that class -- a client only needs two floats, and importing the full job-retry model (with its
|
|
2032
|
+
`retryable_domains` field, which governs WHOLE-JOB attempt retries, a distinct concept from a
|
|
2033
|
+
single outbound delivery's backoff) would be a coupling this module doesn't need.
|
|
2034
|
+
|
|
2035
|
+
`max_attempts` bounds one `_perform_with_retry` call's own retry loop for the classes the
|
|
2036
|
+
contract leaves unbounded (network/5xx/429 -- "spool + backoff", no stated attempt ceiling).
|
|
2037
|
+
STUCK DECISION (fail-safe/reversible, contract silent): capped at a generous default (8) rather
|
|
2038
|
+
than looped forever, because durability already lives in the spool/idempotent-wire-design, not
|
|
2039
|
+
in one blocking call -- a caller that wants to keep trying simply invokes the client method
|
|
2040
|
+
again later (`EventsClient.flush()` is designed to be called repeatedly for exactly this
|
|
2041
|
+
reason). Must stay >= 3 for the 404 rule to ever reach its own `CHANNEL_FAILED` transition
|
|
2042
|
+
within a single call; the default comfortably clears that.
|
|
2043
|
+
"""
|
|
2044
|
+
|
|
2045
|
+
initial_backoff_seconds: float = 1.0
|
|
2046
|
+
max_backoff_seconds: float = 15.0
|
|
2047
|
+
max_attempts: int = 8
|
|
2048
|
+
|
|
2049
|
+
|
|
2050
|
+
def _perform_with_retry(
|
|
2051
|
+
perform: Callable[[int], TransportResponse],
|
|
2052
|
+
*,
|
|
2053
|
+
retry_policy: RetryPolicy,
|
|
2054
|
+
sleep: Callable[[float], None],
|
|
2055
|
+
rng: Callable[[], float] = random.random,
|
|
2056
|
+
deadline: float | None = None,
|
|
2057
|
+
now: Callable[[], float] = time.monotonic,
|
|
2058
|
+
) -> tuple[TransportResponse | None, ChannelError | None]:
|
|
2059
|
+
"""The one retry/backoff engine all three channel clients drive their single HTTP call through.
|
|
2060
|
+
`perform(attempt)` makes ONE attempt (1-based); this loops it, classifies each outcome via
|
|
2061
|
+
`classify_response`, and:
|
|
2062
|
+
|
|
2063
|
+
- returns `(response, None)` once delivered;
|
|
2064
|
+
- raises `HostedFencedError` immediately on 401/403 (never retried, by contract);
|
|
2065
|
+
- raises `HostedChannelFailedError` once a 404 reaches its third attempt;
|
|
2066
|
+
- returns `(response_or_None, error)` for every other terminal outcome (`PERMANENT_ITEM`,
|
|
2067
|
+
`BUDGET_EXCEEDED`) without retrying -- "deterministic rejections are never retried";
|
|
2068
|
+
- otherwise (`RETRYABLE`) sleeps -- honoring a server `Retry-After` over the computed backoff
|
|
2069
|
+
when present -- and tries again, up to `retry_policy.max_attempts`.
|
|
2070
|
+
|
|
2071
|
+
N5/P5: `deadline` (a `time.monotonic()` value, typically the adapter's flush-window end) bounds
|
|
2072
|
+
ATTEMPT SCHEDULING AND SLEEPS ONLY -- no new attempt starts once `now() >= deadline`, and every
|
|
2073
|
+
sleep (computed backoff OR a server `Retry-After`) is clamped to whatever budget remains. It
|
|
2074
|
+
does NOT bound an attempt's own in-flight request: `perform`'s transport call carries its own,
|
|
2075
|
+
separate `timeout` (the caller's second knob -- `Transport.request`'s `timeout` parameter,
|
|
2076
|
+
unrelated to `deadline`), and nothing here clamps that value against the remaining deadline
|
|
2077
|
+
budget. A caller sizing `deadline` against a hard wall-clock guarantee for the whole call is
|
|
2078
|
+
sizing against a guarantee this function does not provide; only "no new attempt starts, and no
|
|
2079
|
+
sleep runs, once the window is gone" is guaranteed. (P5: widening `perform` to accept a clamped
|
|
2080
|
+
per-attempt timeout was considered and is the more complete fix, but at least one caller outside
|
|
2081
|
+
this module -- `hosted_entrypoint.py`'s `ScenariosClient._post`, which calls this function
|
|
2082
|
+
directly with its own single-argument `perform` closure -- is out of this fix's scope, so
|
|
2083
|
+
changing the call signature here would silently break that caller instead of fixing it. This
|
|
2084
|
+
docstring correction is the floor the round-3 review named for exactly that situation.)
|
|
2085
|
+
|
|
2086
|
+
N6: 404 retries are counted SEPARATELY from the overall attempt number (`not_found_attempts`),
|
|
2087
|
+
so an interleaved 5xx (e.g. `503, 503, 404, 404, 404`) does not shorten the 404 budget -- only
|
|
2088
|
+
three OBSERVED 404s reach `classify_response`'s `CHANNEL_FAILED` transition, regardless of what
|
|
2089
|
+
else happened in between.
|
|
2090
|
+
"""
|
|
2091
|
+
attempt = 0
|
|
2092
|
+
not_found_attempts = 0
|
|
2093
|
+
while True:
|
|
2094
|
+
attempt += 1
|
|
2095
|
+
if deadline is not None and now() >= deadline:
|
|
2096
|
+
return None, ChannelError(
|
|
2097
|
+
ChannelOutcome.RETRYABLE,
|
|
2098
|
+
FailureDomain.CONNECTIVITY,
|
|
2099
|
+
"deadline_exceeded",
|
|
2100
|
+
"the flush-window deadline elapsed before this attempt could be made",
|
|
2101
|
+
)
|
|
2102
|
+
try:
|
|
2103
|
+
response = perform(attempt)
|
|
2104
|
+
except TransportError:
|
|
2105
|
+
error = classify_response(None, None, attempt=attempt)
|
|
2106
|
+
response = None
|
|
2107
|
+
else:
|
|
2108
|
+
status = response.status_code
|
|
2109
|
+
if status == 404:
|
|
2110
|
+
not_found_attempts += 1
|
|
2111
|
+
error = classify_response(
|
|
2112
|
+
status,
|
|
2113
|
+
response.body,
|
|
2114
|
+
attempt=not_found_attempts if status == 404 else attempt,
|
|
2115
|
+
retry_after_seconds=_parse_retry_after(response.headers),
|
|
2116
|
+
)
|
|
2117
|
+
if error is None:
|
|
2118
|
+
return response, None
|
|
2119
|
+
if error.outcome is ChannelOutcome.FENCED:
|
|
2120
|
+
raise HostedFencedError(error)
|
|
2121
|
+
if error.outcome is ChannelOutcome.CHANNEL_FAILED:
|
|
2122
|
+
raise HostedChannelFailedError(error)
|
|
2123
|
+
if error.outcome is not ChannelOutcome.RETRYABLE:
|
|
2124
|
+
return response, error
|
|
2125
|
+
if attempt >= retry_policy.max_attempts:
|
|
2126
|
+
return response, error
|
|
2127
|
+
delay = error.retry_after_seconds
|
|
2128
|
+
if delay is not None:
|
|
2129
|
+
delay = min(
|
|
2130
|
+
delay, retry_policy.max_backoff_seconds
|
|
2131
|
+
) # N5: clamp Retry-After
|
|
2132
|
+
else:
|
|
2133
|
+
delay = compute_backoff_seconds(
|
|
2134
|
+
attempt,
|
|
2135
|
+
initial_backoff_seconds=retry_policy.initial_backoff_seconds,
|
|
2136
|
+
max_backoff_seconds=retry_policy.max_backoff_seconds,
|
|
2137
|
+
rng=rng,
|
|
2138
|
+
)
|
|
2139
|
+
if deadline is not None:
|
|
2140
|
+
remaining = deadline - now()
|
|
2141
|
+
if remaining <= 0:
|
|
2142
|
+
return response, error
|
|
2143
|
+
delay = min(delay, remaining)
|
|
2144
|
+
sleep(delay)
|
|
2145
|
+
|
|
2146
|
+
|
|
2147
|
+
# =================================================================================================
|
|
2148
|
+
# Channel 1 -- Events client. Batches spooled events, advances the watermark only on confirmed
|
|
2149
|
+
# delivery.
|
|
2150
|
+
# =================================================================================================
|
|
2151
|
+
|
|
2152
|
+
|
|
2153
|
+
@dataclass(frozen=True)
|
|
2154
|
+
class EventsFlushResult:
|
|
2155
|
+
delivered_count: int
|
|
2156
|
+
acked_through_sequence: int | None
|
|
2157
|
+
rejected: list[dict[str, Any]]
|
|
2158
|
+
error: ChannelError | None
|
|
2159
|
+
# v1.3 (M7): set when the platform's `acked_through_sequence` fell outside the spool's trusted
|
|
2160
|
+
# range and was ignored rather than trusted -- `error` stays `None` because the HTTP delivery
|
|
2161
|
+
# itself succeeded; only the ack body was untrustworthy.
|
|
2162
|
+
ack_out_of_range: bool = False
|
|
2163
|
+
# N2/N3: set when a 2xx response's `acked_through_sequence` was missing, `null`, or not an
|
|
2164
|
+
# int -- a protocol violation distinct from "present but out of range" above. `error` stays
|
|
2165
|
+
# `None` for the same reason: the HTTP delivery itself succeeded.
|
|
2166
|
+
ack_missing: bool = False
|
|
2167
|
+
# P9: the SpooledRecord bodies dropped this call (a subset of `batch`, keyed by `rejected`),
|
|
2168
|
+
# captured BEFORE `drop_many` removes them from disk. The contract requires a rejected event's
|
|
2169
|
+
# payload be "written to the artifact spool (`log` kind)" -- without this, a caller has no way
|
|
2170
|
+
# to recover that payload at all once `flush()` returns, since the cap applied inside `flush()`
|
|
2171
|
+
# makes it impossible to reliably re-derive which spooled records were even in this batch.
|
|
2172
|
+
dropped_records: list[SpooledRecord] = field(default_factory=list)
|
|
2173
|
+
|
|
2174
|
+
|
|
2175
|
+
_EVENTS_BATCH_PREFIX = (
|
|
2176
|
+
b'{"schema_version":"' + EVENT_SCHEMA_VERSION.encode("utf-8") + b'","events":['
|
|
2177
|
+
)
|
|
2178
|
+
_EVENTS_BATCH_SUFFIX = b"]}"
|
|
2179
|
+
|
|
2180
|
+
|
|
2181
|
+
def _encode_events_batch(records: list[SpooledRecord]) -> bytes:
|
|
2182
|
+
"""N19: the contract says "serialize once, spool the bytes, re-send verbatim; never
|
|
2183
|
+
re-serialize on retry" -- for the events BATCH ENVELOPE itself, not just each event inside it.
|
|
2184
|
+
Handing a decoded dict to `json_body=` (the previous shape) let `requests` re-serialize the
|
|
2185
|
+
envelope with its own settings on every send; this instead concatenates the exact bytes
|
|
2186
|
+
`OutboundSpool.append` already wrote for each event, closing the deviation rather than merely
|
|
2187
|
+
documenting it. Safe because `EVENT_SCHEMA_VERSION` is a fixed ASCII constant with no bytes
|
|
2188
|
+
needing escape."""
|
|
2189
|
+
return (
|
|
2190
|
+
_EVENTS_BATCH_PREFIX
|
|
2191
|
+
+ b",".join(record.body for record in records)
|
|
2192
|
+
+ _EVENTS_BATCH_SUFFIX
|
|
2193
|
+
)
|
|
2194
|
+
|
|
2195
|
+
|
|
2196
|
+
class EventsClient:
|
|
2197
|
+
"""Delivers `OutboundSpool`-backed Channel 1 events to `endpoints.events`. One `flush()` call
|
|
2198
|
+
sends one batch (<= `EVENTS_MAX_BATCH` events, <= `EVENTS_MAX_BATCH_BYTES`) of everything
|
|
2199
|
+
spooled since the last confirmed watermark, in spool order
|
|
2200
|
+
(`OutboundSpool.pending_since_watermark`, which reads the log in append/sequence order -- the
|
|
2201
|
+
platform's own ordering rule, "`(attempt_number, sequence)`", so a single-attempt process
|
|
2202
|
+
satisfies it for free). Re-sends spooled bytes verbatim (N19, `_encode_events_batch`) -- never
|
|
2203
|
+
recomputing an event's own `digest`, so "serialize once ... never re-serialize on retry" holds
|
|
2204
|
+
for the one thing that must never drift (the per-event digest, embedded as data).
|
|
2205
|
+
|
|
2206
|
+
`flush()` is meant to be called repeatedly (by whatever background loop owns the call cadence,
|
|
2207
|
+
a scheduler concern outside this module) -- each call is a complete, self-contained delivery
|
|
2208
|
+
attempt (with its own internal retry/backoff via `_perform_with_retry`) that advances the
|
|
2209
|
+
watermark exactly as far as the platform confirmed and leaves everything else spooled for the
|
|
2210
|
+
next call.
|
|
2211
|
+
|
|
2212
|
+
The ack body is UNTRUSTED PLATFORM INPUT end to end (v1.3): `acked_through_sequence` goes
|
|
2213
|
+
through the spool's own M7 clamp (`advance_watermark`); `rejected[]` is filtered to sequences
|
|
2214
|
+
actually present in the batch just sent BEFORE anything is done with it (N1) -- a value the
|
|
2215
|
+
guest never sent cannot cause a drop, and dropping never itself advances the watermark (see
|
|
2216
|
+
`OutboundSpool.drop_many`) -- so the batch-level `advance_watermark(acked_through_sequence)` is
|
|
2217
|
+
the single chokepoint either way.
|
|
2218
|
+
"""
|
|
2219
|
+
|
|
2220
|
+
def __init__(
|
|
2221
|
+
self,
|
|
2222
|
+
capabilities: HostedCapabilities,
|
|
2223
|
+
spool: OutboundSpool,
|
|
2224
|
+
transport: Transport | None = None,
|
|
2225
|
+
*,
|
|
2226
|
+
retry_policy: RetryPolicy | None = None,
|
|
2227
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
2228
|
+
rng: Callable[[], float] = random.random,
|
|
2229
|
+
batch_size: int = EVENTS_MAX_BATCH,
|
|
2230
|
+
channel_state: ChannelState | None = None,
|
|
2231
|
+
) -> None:
|
|
2232
|
+
self._capabilities = capabilities
|
|
2233
|
+
self._spool = spool
|
|
2234
|
+
self._transport = transport or RequestsTransport()
|
|
2235
|
+
self._retry_policy = retry_policy or RetryPolicy()
|
|
2236
|
+
self._sleep = sleep
|
|
2237
|
+
self._rng = rng
|
|
2238
|
+
self._batch_size = max(1, min(batch_size, EVENTS_MAX_BATCH))
|
|
2239
|
+
self._channel_state = channel_state or ChannelState()
|
|
2240
|
+
|
|
2241
|
+
def _cap_batch(self, records: list[SpooledRecord]) -> list[SpooledRecord]:
|
|
2242
|
+
"""Proactive half of N7: cap by event count (`_batch_size`) AND cumulative canonical bytes
|
|
2243
|
+
(`EVENTS_MAX_BATCH_BYTES`) before ever building a request -- reduces how often the reactive
|
|
2244
|
+
413-halving in `flush()` below is ever needed. Always includes at least one record (its own
|
|
2245
|
+
oversized payload is HostedEventDraft's problem, at spool-append time, not this cap's)."""
|
|
2246
|
+
capped = records[: self._batch_size]
|
|
2247
|
+
limited: list[SpooledRecord] = []
|
|
2248
|
+
total_bytes = 0
|
|
2249
|
+
for record in capped:
|
|
2250
|
+
if limited and total_bytes + len(record.body) > EVENTS_MAX_BATCH_BYTES:
|
|
2251
|
+
break
|
|
2252
|
+
limited.append(record)
|
|
2253
|
+
total_bytes += len(record.body)
|
|
2254
|
+
return limited
|
|
2255
|
+
|
|
2256
|
+
def flush(self, *, deadline: float | None = None) -> EventsFlushResult:
|
|
2257
|
+
self._channel_state.check()
|
|
2258
|
+
batch = self._cap_batch(self._spool.pending_since_watermark())
|
|
2259
|
+
if not batch:
|
|
2260
|
+
return EventsFlushResult(
|
|
2261
|
+
delivered_count=0,
|
|
2262
|
+
acked_through_sequence=self._spool.watermark(),
|
|
2263
|
+
rejected=[],
|
|
2264
|
+
error=None,
|
|
2265
|
+
)
|
|
2266
|
+
|
|
2267
|
+
response: TransportResponse | None = None
|
|
2268
|
+
error: ChannelError | None = None
|
|
2269
|
+
while True:
|
|
2270
|
+
body_bytes = _encode_events_batch(batch)
|
|
2271
|
+
headers = {
|
|
2272
|
+
**self._capabilities.auth_headers(),
|
|
2273
|
+
"Content-Type": "application/json",
|
|
2274
|
+
}
|
|
2275
|
+
|
|
2276
|
+
def perform(_attempt: int) -> TransportResponse:
|
|
2277
|
+
return self._transport.request(
|
|
2278
|
+
"POST",
|
|
2279
|
+
self._capabilities.endpoints.events,
|
|
2280
|
+
headers=headers,
|
|
2281
|
+
data=body_bytes,
|
|
2282
|
+
)
|
|
2283
|
+
|
|
2284
|
+
try:
|
|
2285
|
+
response, error = _perform_with_retry(
|
|
2286
|
+
perform,
|
|
2287
|
+
retry_policy=self._retry_policy,
|
|
2288
|
+
sleep=self._sleep,
|
|
2289
|
+
rng=self._rng,
|
|
2290
|
+
deadline=deadline,
|
|
2291
|
+
)
|
|
2292
|
+
except (HostedFencedError, HostedChannelFailedError) as exc:
|
|
2293
|
+
self._channel_state.latch(exc)
|
|
2294
|
+
raise
|
|
2295
|
+
# N7: a 413 on the events channel -- reactively halve the batch and try again rather
|
|
2296
|
+
# than returning a permanent, non-progressing error for the whole thing. Stops once a
|
|
2297
|
+
# single event alone still 413s (defensive; should not happen under EVENT_PAYLOAD_MAX_BYTES).
|
|
2298
|
+
if error is not None and error.status_code == 413 and len(batch) > 1:
|
|
2299
|
+
logger.warning(
|
|
2300
|
+
"events flush: batch of %d events (%d bytes) was rejected with 413 -- halving "
|
|
2301
|
+
"and retrying",
|
|
2302
|
+
len(batch),
|
|
2303
|
+
len(body_bytes),
|
|
2304
|
+
)
|
|
2305
|
+
batch = batch[: len(batch) // 2]
|
|
2306
|
+
continue
|
|
2307
|
+
break
|
|
2308
|
+
|
|
2309
|
+
if error is not None or response is None:
|
|
2310
|
+
return EventsFlushResult(
|
|
2311
|
+
delivered_count=0, acked_through_sequence=None, rejected=[], error=error
|
|
2312
|
+
)
|
|
2313
|
+
|
|
2314
|
+
# N2/N3: the ack body is untrusted platform input, defensively parsed -- a wrong type or a
|
|
2315
|
+
# missing key must never raise out of flush() (that would silence the flusher loop, B1's
|
|
2316
|
+
# outcome by a different route) and must never be treated as "0" (that would silently
|
|
2317
|
+
# re-send the same batch forever, N3).
|
|
2318
|
+
body = response.body if isinstance(response.body, dict) else {}
|
|
2319
|
+
raw_acked = body.get("acked_through_sequence")
|
|
2320
|
+
acked_through: int | None
|
|
2321
|
+
if isinstance(raw_acked, int) and not isinstance(raw_acked, bool):
|
|
2322
|
+
acked_through = raw_acked
|
|
2323
|
+
else:
|
|
2324
|
+
acked_through = None
|
|
2325
|
+
logger.warning(
|
|
2326
|
+
"events flush: 2xx response has a missing/invalid acked_through_sequence (got %r) "
|
|
2327
|
+
"-- treating as a protocol violation, not advancing the watermark",
|
|
2328
|
+
raw_acked,
|
|
2329
|
+
)
|
|
2330
|
+
|
|
2331
|
+
raw_rejected = body.get("rejected")
|
|
2332
|
+
if isinstance(raw_rejected, list):
|
|
2333
|
+
rejected_entries = [
|
|
2334
|
+
entry for entry in raw_rejected if isinstance(entry, dict)
|
|
2335
|
+
]
|
|
2336
|
+
if len(rejected_entries) != len(raw_rejected):
|
|
2337
|
+
logger.warning(
|
|
2338
|
+
"events flush: rejected[] contained non-object entries -- ignoring them"
|
|
2339
|
+
)
|
|
2340
|
+
else:
|
|
2341
|
+
rejected_entries = []
|
|
2342
|
+
if raw_rejected is not None:
|
|
2343
|
+
logger.warning(
|
|
2344
|
+
"events flush: rejected is %r, not a list -- treating as empty",
|
|
2345
|
+
type(raw_rejected).__name__,
|
|
2346
|
+
)
|
|
2347
|
+
|
|
2348
|
+
if acked_through is None:
|
|
2349
|
+
return EventsFlushResult(
|
|
2350
|
+
delivered_count=0,
|
|
2351
|
+
acked_through_sequence=self._spool.watermark(),
|
|
2352
|
+
rejected=[],
|
|
2353
|
+
error=None,
|
|
2354
|
+
ack_missing=True,
|
|
2355
|
+
)
|
|
2356
|
+
|
|
2357
|
+
# N1: filter rejected[] to sequences the guest ACTUALLY sent in this batch, before doing
|
|
2358
|
+
# anything with them -- a sequence the platform names that was never in `batch` is
|
|
2359
|
+
# untrusted input this module owes no obedience to (it cannot be dropped, since it was
|
|
2360
|
+
# never spooled under that number in the first place, and trusting it would let a
|
|
2361
|
+
# malformed ack orphan pending records by a route the M7 clamp doesn't guard).
|
|
2362
|
+
batch_sequences = {
|
|
2363
|
+
record.sequence for record in batch if record.sequence is not None
|
|
2364
|
+
}
|
|
2365
|
+
valid_rejected: list[dict[str, Any]] = []
|
|
2366
|
+
for entry in rejected_entries:
|
|
2367
|
+
sequence = entry.get("sequence")
|
|
2368
|
+
if (
|
|
2369
|
+
isinstance(sequence, int)
|
|
2370
|
+
and not isinstance(sequence, bool)
|
|
2371
|
+
and sequence in batch_sequences
|
|
2372
|
+
):
|
|
2373
|
+
valid_rejected.append(entry)
|
|
2374
|
+
else:
|
|
2375
|
+
logger.warning(
|
|
2376
|
+
"events flush: rejected entry names sequence=%r, which was not sent in this "
|
|
2377
|
+
"batch (sent=%s) -- ignoring as untrusted platform input",
|
|
2378
|
+
sequence,
|
|
2379
|
+
sorted(batch_sequences),
|
|
2380
|
+
)
|
|
2381
|
+
|
|
2382
|
+
# P9: capture the dropped records' own bodies BEFORE drop_many physically removes them --
|
|
2383
|
+
# once removed, this is the only place a caller can still recover the payload the contract
|
|
2384
|
+
# requires be "written to the artifact spool (`log` kind)" for a rejected event.
|
|
2385
|
+
rejected_sequences = {entry["sequence"] for entry in valid_rejected}
|
|
2386
|
+
dropped_records = [
|
|
2387
|
+
record for record in batch if record.sequence in rejected_sequences
|
|
2388
|
+
]
|
|
2389
|
+
|
|
2390
|
+
# N1/N14: pure physical removal, batched into one rewrite -- drop_many never touches the
|
|
2391
|
+
# watermark; the batch-level advance_watermark(acked_through) below is the ONLY chokepoint.
|
|
2392
|
+
self._spool.drop_many(rejected_sequences)
|
|
2393
|
+
|
|
2394
|
+
try:
|
|
2395
|
+
# "the watermark is highest-processed -- accepted AND rejected sequences both advance
|
|
2396
|
+
# it." v1.3: `acked_through_sequence` is untrusted input -- the spool itself enforces
|
|
2397
|
+
# the clamp (M7).
|
|
2398
|
+
self._spool.advance_watermark(acked_through)
|
|
2399
|
+
except OutboundSpoolError:
|
|
2400
|
+
logger.warning(
|
|
2401
|
+
"events flush: platform returned an untrusted acked_through_sequence=%s outside "
|
|
2402
|
+
"the guest's trusted range -- ignoring the ack, watermark unchanged at %s",
|
|
2403
|
+
acked_through,
|
|
2404
|
+
self._spool.watermark(),
|
|
2405
|
+
)
|
|
2406
|
+
return EventsFlushResult(
|
|
2407
|
+
delivered_count=0,
|
|
2408
|
+
acked_through_sequence=self._spool.watermark(),
|
|
2409
|
+
rejected=[],
|
|
2410
|
+
error=None,
|
|
2411
|
+
ack_out_of_range=True,
|
|
2412
|
+
dropped_records=dropped_records,
|
|
2413
|
+
)
|
|
2414
|
+
|
|
2415
|
+
delivered_count = sum(
|
|
2416
|
+
1
|
|
2417
|
+
for record in batch
|
|
2418
|
+
if record.sequence is not None
|
|
2419
|
+
and record.sequence <= acked_through
|
|
2420
|
+
and record.sequence not in rejected_sequences
|
|
2421
|
+
)
|
|
2422
|
+
return EventsFlushResult(
|
|
2423
|
+
delivered_count=delivered_count,
|
|
2424
|
+
acked_through_sequence=acked_through,
|
|
2425
|
+
rejected=valid_rejected,
|
|
2426
|
+
error=None,
|
|
2427
|
+
dropped_records=dropped_records,
|
|
2428
|
+
)
|
|
2429
|
+
|
|
2430
|
+
|
|
2431
|
+
# =================================================================================================
|
|
2432
|
+
# Channel 2 -- Result receipts. Typed `ResultReceiptDraft` + delivery.
|
|
2433
|
+
# =================================================================================================
|
|
2434
|
+
|
|
2435
|
+
|
|
2436
|
+
class ScenarioStatus(str, Enum):
|
|
2437
|
+
PASSED = "passed"
|
|
2438
|
+
FAILED = "failed"
|
|
2439
|
+
ERRORED = "errored"
|
|
2440
|
+
SKIPPED = "skipped"
|
|
2441
|
+
|
|
2442
|
+
|
|
2443
|
+
class SubGoalResult(BaseModel):
|
|
2444
|
+
model_config = ConfigDict(extra="forbid")
|
|
2445
|
+
|
|
2446
|
+
name: str = Field(min_length=1)
|
|
2447
|
+
held: bool | None
|
|
2448
|
+
reason: str | None
|
|
2449
|
+
judged: bool
|
|
2450
|
+
|
|
2451
|
+
|
|
2452
|
+
class MetricEvaluation(BaseModel):
|
|
2453
|
+
model_config = ConfigDict(extra="forbid")
|
|
2454
|
+
|
|
2455
|
+
name: str = Field(min_length=1)
|
|
2456
|
+
kind: Literal["metric"]
|
|
2457
|
+
score: float = Field(ge=0.0, le=1.0)
|
|
2458
|
+
reason: str
|
|
2459
|
+
|
|
2460
|
+
|
|
2461
|
+
class CheckpointEvaluation(BaseModel):
|
|
2462
|
+
model_config = ConfigDict(extra="forbid")
|
|
2463
|
+
|
|
2464
|
+
name: str = Field(min_length=1)
|
|
2465
|
+
kind: Literal["checkpoint"]
|
|
2466
|
+
passed: bool
|
|
2467
|
+
reason: str
|
|
2468
|
+
|
|
2469
|
+
|
|
2470
|
+
EvaluationResult = MetricEvaluation | CheckpointEvaluation
|
|
2471
|
+
|
|
2472
|
+
|
|
2473
|
+
class CallSummary(BaseModel):
|
|
2474
|
+
model_config = ConfigDict(extra="forbid")
|
|
2475
|
+
|
|
2476
|
+
started_at: str
|
|
2477
|
+
ended_at: str
|
|
2478
|
+
duration_ms: int = Field(ge=0)
|
|
2479
|
+
turns: int = Field(ge=0)
|
|
2480
|
+
transcript_artifact: str | None
|
|
2481
|
+
recording_artifacts: list[str] = Field(default_factory=list)
|
|
2482
|
+
stop_reason: str | None = None
|
|
2483
|
+
|
|
2484
|
+
@model_validator(mode="after")
|
|
2485
|
+
def _validate(self) -> "CallSummary":
|
|
2486
|
+
for label, value in (
|
|
2487
|
+
("started_at", self.started_at),
|
|
2488
|
+
("ended_at", self.ended_at),
|
|
2489
|
+
):
|
|
2490
|
+
if not is_valid_rfc3339_millis(value):
|
|
2491
|
+
raise ValueError(f"call_timestamp_invalid: {label}={value!r}")
|
|
2492
|
+
if self.transcript_artifact is not None and not is_valid_digest(
|
|
2493
|
+
self.transcript_artifact
|
|
2494
|
+
):
|
|
2495
|
+
raise ValueError(
|
|
2496
|
+
f"call_transcript_artifact_invalid: {self.transcript_artifact!r}"
|
|
2497
|
+
)
|
|
2498
|
+
for artifact in self.recording_artifacts:
|
|
2499
|
+
if not is_valid_digest(artifact):
|
|
2500
|
+
raise ValueError(f"call_recording_artifact_invalid: {artifact!r}")
|
|
2501
|
+
return self
|
|
2502
|
+
|
|
2503
|
+
|
|
2504
|
+
def _unset_default_fields(model: BaseModel, prefix: str = "") -> list[str]:
|
|
2505
|
+
"""N23: names the fields a caller did NOT explicitly set (filled from a pydantic default) --
|
|
2506
|
+
the actual shape of a digest-mismatch bug like the review's example: a raw `call` dict that
|
|
2507
|
+
omits `recording_artifacts` computes an external digest over an object without that key, while
|
|
2508
|
+
the model's own re-derivation fills in `recording_artifacts: []`. This is not a general diff
|
|
2509
|
+
against the caller's original raw dict (a model validator has no access to that, only to what
|
|
2510
|
+
pydantic recorded via `model_fields_set`) -- it is the honest, dotted-path subset available from
|
|
2511
|
+
inside the model: which fields with defaults were left unset, one level into nested models
|
|
2512
|
+
(covers `call.recording_artifacts`, not just top-level `world_index`/`schema_version`)."""
|
|
2513
|
+
names: list[str] = []
|
|
2514
|
+
for name in type(model).model_fields:
|
|
2515
|
+
if name == "digest":
|
|
2516
|
+
continue
|
|
2517
|
+
path = f"{prefix}{name}"
|
|
2518
|
+
if name not in model.model_fields_set:
|
|
2519
|
+
names.append(path)
|
|
2520
|
+
value = getattr(model, name)
|
|
2521
|
+
if isinstance(value, BaseModel):
|
|
2522
|
+
names.extend(_unset_default_fields(value, f"{path}."))
|
|
2523
|
+
return names
|
|
2524
|
+
|
|
2525
|
+
|
|
2526
|
+
class ResultReceiptDraft(BaseModel):
|
|
2527
|
+
"""Channel 2's wire shape. Mirrors `HostedEventDraft`'s pattern: the caller supplies `digest`
|
|
2528
|
+
(via `build_result_receipt`, computed with `whole_object_digest`), the model re-derives and
|
|
2529
|
+
rejects a mismatch, plus the two exact-shape rules the contract states as literal requirements
|
|
2530
|
+
rather than general validation ("`skipped` receipt body (exact)" and "`errored` receipt body").
|
|
2531
|
+
"""
|
|
2532
|
+
|
|
2533
|
+
model_config = ConfigDict(extra="forbid")
|
|
2534
|
+
|
|
2535
|
+
schema_version: str = RESULT_SCHEMA_VERSION
|
|
2536
|
+
job_id: str = Field(min_length=1)
|
|
2537
|
+
attempt_id: str = Field(min_length=1)
|
|
2538
|
+
attempt_number: int = Field(ge=1)
|
|
2539
|
+
scenario_key: str = Field(min_length=1)
|
|
2540
|
+
scenario_id: str = Field(min_length=1)
|
|
2541
|
+
scenario_attempt: Literal[1, 2]
|
|
2542
|
+
world_index: int | None = Field(default=None, ge=0)
|
|
2543
|
+
status: ScenarioStatus
|
|
2544
|
+
sub_goals: list[SubGoalResult]
|
|
2545
|
+
evaluations: list[EvaluationResult]
|
|
2546
|
+
call: CallSummary | None
|
|
2547
|
+
failure: TerminalFailure | None
|
|
2548
|
+
digest: str
|
|
2549
|
+
|
|
2550
|
+
@model_validator(mode="after")
|
|
2551
|
+
def _validate(self) -> "ResultReceiptDraft":
|
|
2552
|
+
if self.schema_version != RESULT_SCHEMA_VERSION:
|
|
2553
|
+
raise ValueError(f"result_schema_unsupported: {self.schema_version}")
|
|
2554
|
+
if not is_valid_digest(self.digest):
|
|
2555
|
+
raise ValueError(f"receipt_digest_invalid: {self.digest!r}")
|
|
2556
|
+
expected_body = self.model_dump(mode="json", exclude={"digest"})
|
|
2557
|
+
# ``stop_reason`` was added after the initial receipt protocol. Preserve
|
|
2558
|
+
# byte-for-byte compatibility for callers that omit it, while including
|
|
2559
|
+
# it in both the digest and wire body whenever it is explicitly supplied.
|
|
2560
|
+
if self.call is not None and "stop_reason" not in self.call.model_fields_set:
|
|
2561
|
+
expected_body["call"].pop("stop_reason", None)
|
|
2562
|
+
expected = whole_object_digest(expected_body)
|
|
2563
|
+
if self.digest != expected:
|
|
2564
|
+
unset = _unset_default_fields(self)
|
|
2565
|
+
hint = (
|
|
2566
|
+
f" -- fields not explicitly set, filled from defaults: {', '.join(unset)}"
|
|
2567
|
+
if unset
|
|
2568
|
+
else ""
|
|
2569
|
+
)
|
|
2570
|
+
raise ValueError(f"receipt_digest_mismatch{hint}")
|
|
2571
|
+
|
|
2572
|
+
if self.status is ScenarioStatus.SKIPPED:
|
|
2573
|
+
if (
|
|
2574
|
+
self.scenario_attempt != 1
|
|
2575
|
+
or self.world_index is not None
|
|
2576
|
+
or self.sub_goals
|
|
2577
|
+
or self.evaluations
|
|
2578
|
+
or self.call is not None
|
|
2579
|
+
or self.failure is not None
|
|
2580
|
+
):
|
|
2581
|
+
raise ValueError("skipped_receipt_shape_invalid")
|
|
2582
|
+
elif self.status is ScenarioStatus.ERRORED and self.failure is None:
|
|
2583
|
+
raise ValueError("errored_receipt_requires_failure")
|
|
2584
|
+
return self
|
|
2585
|
+
|
|
2586
|
+
|
|
2587
|
+
def build_result_receipt(
|
|
2588
|
+
*,
|
|
2589
|
+
job_id: str,
|
|
2590
|
+
attempt_id: str,
|
|
2591
|
+
attempt_number: int,
|
|
2592
|
+
scenario_key: str,
|
|
2593
|
+
scenario_id: str,
|
|
2594
|
+
scenario_attempt: Literal[1, 2],
|
|
2595
|
+
world_index: int | None,
|
|
2596
|
+
status: ScenarioStatus,
|
|
2597
|
+
sub_goals: list[dict[str, Any]],
|
|
2598
|
+
evaluations: list[dict[str, Any]],
|
|
2599
|
+
call: dict[str, Any] | None,
|
|
2600
|
+
failure: dict[str, Any] | None,
|
|
2601
|
+
extra_secret_values: tuple[str, ...] = (),
|
|
2602
|
+
) -> dict[str, Any]:
|
|
2603
|
+
"""Validate one receipt's shape and compute its digest, returning a plain wire-ready dict.
|
|
2604
|
+
Mirrors `build_event_record`'s contract exactly: callers must pass already wire-typed values
|
|
2605
|
+
(e.g. a metric `score` as `float`, never `int`) -- the digest is computed on the RAW input
|
|
2606
|
+
before model validation/coercion, so a type looseness here fails loudly as a digest mismatch
|
|
2607
|
+
rather than silently spooling a digest that doesn't match what gets sent.
|
|
2608
|
+
|
|
2609
|
+
N9: `redact_outbound_text` runs on `sub_goals[].reason`, `evaluations[].reason`, and
|
|
2610
|
+
`failure.{code,message}` (P8) BEFORE the digest is computed -- same ordering rationale as
|
|
2611
|
+
`build_event_record`: the embedded digest must match the redacted bytes actually sent.
|
|
2612
|
+
"""
|
|
2613
|
+
|
|
2614
|
+
def _redact(text: Any) -> Any:
|
|
2615
|
+
return (
|
|
2616
|
+
redact_outbound_text(text, extra_secret_values)
|
|
2617
|
+
if isinstance(text, str)
|
|
2618
|
+
else text
|
|
2619
|
+
)
|
|
2620
|
+
|
|
2621
|
+
sub_goals = [
|
|
2622
|
+
{**goal, "reason": _redact(goal.get("reason"))}
|
|
2623
|
+
if isinstance(goal, dict)
|
|
2624
|
+
else goal
|
|
2625
|
+
for goal in sub_goals
|
|
2626
|
+
]
|
|
2627
|
+
evaluations = [
|
|
2628
|
+
{**item, "reason": _redact(item.get("reason"))}
|
|
2629
|
+
if isinstance(item, dict)
|
|
2630
|
+
else item
|
|
2631
|
+
for item in evaluations
|
|
2632
|
+
]
|
|
2633
|
+
if isinstance(failure, dict):
|
|
2634
|
+
updates = {
|
|
2635
|
+
key: _redact(failure[key])
|
|
2636
|
+
for key in ("code", "message")
|
|
2637
|
+
if isinstance(failure.get(key), str)
|
|
2638
|
+
}
|
|
2639
|
+
if updates:
|
|
2640
|
+
failure = {**failure, **updates}
|
|
2641
|
+
|
|
2642
|
+
core: dict[str, Any] = {
|
|
2643
|
+
"schema_version": RESULT_SCHEMA_VERSION,
|
|
2644
|
+
"job_id": job_id,
|
|
2645
|
+
"attempt_id": attempt_id,
|
|
2646
|
+
"attempt_number": attempt_number,
|
|
2647
|
+
"scenario_key": scenario_key,
|
|
2648
|
+
"scenario_id": scenario_id,
|
|
2649
|
+
"scenario_attempt": scenario_attempt,
|
|
2650
|
+
"world_index": world_index,
|
|
2651
|
+
"status": status.value if isinstance(status, ScenarioStatus) else status,
|
|
2652
|
+
"sub_goals": sub_goals,
|
|
2653
|
+
"evaluations": evaluations,
|
|
2654
|
+
"call": call,
|
|
2655
|
+
"failure": failure,
|
|
2656
|
+
}
|
|
2657
|
+
digest = whole_object_digest(core)
|
|
2658
|
+
draft = ResultReceiptDraft.model_validate({**core, "digest": digest})
|
|
2659
|
+
wire = draft.model_dump(mode="json")
|
|
2660
|
+
if call is not None and "stop_reason" not in call:
|
|
2661
|
+
wire["call"].pop("stop_reason", None)
|
|
2662
|
+
return wire
|
|
2663
|
+
|
|
2664
|
+
|
|
2665
|
+
def build_skipped_receipt(
|
|
2666
|
+
*,
|
|
2667
|
+
job_id: str,
|
|
2668
|
+
attempt_id: str,
|
|
2669
|
+
attempt_number: int,
|
|
2670
|
+
scenario_key: str,
|
|
2671
|
+
scenario_id: str,
|
|
2672
|
+
) -> dict[str, Any]:
|
|
2673
|
+
"""The "exact" synthesized body for a scenario that never ran ("The guest synthesizes these
|
|
2674
|
+
during the flush window; the finalizer backfills any still missing")."""
|
|
2675
|
+
return build_result_receipt(
|
|
2676
|
+
job_id=job_id,
|
|
2677
|
+
attempt_id=attempt_id,
|
|
2678
|
+
attempt_number=attempt_number,
|
|
2679
|
+
scenario_key=scenario_key,
|
|
2680
|
+
scenario_id=scenario_id,
|
|
2681
|
+
scenario_attempt=1,
|
|
2682
|
+
world_index=None,
|
|
2683
|
+
status=ScenarioStatus.SKIPPED,
|
|
2684
|
+
sub_goals=[],
|
|
2685
|
+
evaluations=[],
|
|
2686
|
+
call=None,
|
|
2687
|
+
failure=None,
|
|
2688
|
+
)
|
|
2689
|
+
|
|
2690
|
+
|
|
2691
|
+
@dataclass(frozen=True)
|
|
2692
|
+
class ReceiptPushResult:
|
|
2693
|
+
delivered: bool
|
|
2694
|
+
already_existed: bool
|
|
2695
|
+
error: ChannelError | None
|
|
2696
|
+
|
|
2697
|
+
|
|
2698
|
+
class ResultsClient:
|
|
2699
|
+
"""Delivers one Channel 2 receipt per call to `endpoints.results`. No spool/watermark of its
|
|
2700
|
+
own -- unlike events, receipts carry no `sequence`; their idempotency key is `(job_id,
|
|
2701
|
+
scenario_key)` (contract), so redelivery safety comes from the wire protocol itself (`200`
|
|
2702
|
+
duplicate on a matching digest) rather than from a local ack cursor. A caller that wants
|
|
2703
|
+
durable at-least-once delivery across a process crash owns that queuing (e.g. an
|
|
2704
|
+
`OutboundSpool(sequenced=False)`, exposed by this module for exactly this) and simply calls
|
|
2705
|
+
`push()` again for anything not yet confirmed -- safe because the platform's own idempotency
|
|
2706
|
+
check is what makes a redelivery a no-op, not any state this client keeps.
|
|
2707
|
+
"""
|
|
2708
|
+
|
|
2709
|
+
def __init__(
|
|
2710
|
+
self,
|
|
2711
|
+
capabilities: HostedCapabilities,
|
|
2712
|
+
transport: Transport | None = None,
|
|
2713
|
+
*,
|
|
2714
|
+
retry_policy: RetryPolicy | None = None,
|
|
2715
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
2716
|
+
rng: Callable[[], float] = random.random,
|
|
2717
|
+
channel_state: ChannelState | None = None,
|
|
2718
|
+
) -> None:
|
|
2719
|
+
self._capabilities = capabilities
|
|
2720
|
+
self._transport = transport or RequestsTransport()
|
|
2721
|
+
self._retry_policy = retry_policy or RetryPolicy()
|
|
2722
|
+
self._sleep = sleep
|
|
2723
|
+
self._rng = rng
|
|
2724
|
+
self._channel_state = channel_state or ChannelState()
|
|
2725
|
+
|
|
2726
|
+
def push(
|
|
2727
|
+
self, receipt: dict[str, Any], *, deadline: float | None = None
|
|
2728
|
+
) -> ReceiptPushResult:
|
|
2729
|
+
self._channel_state.check()
|
|
2730
|
+
|
|
2731
|
+
def perform(_attempt: int) -> TransportResponse:
|
|
2732
|
+
return self._transport.request(
|
|
2733
|
+
"POST",
|
|
2734
|
+
self._capabilities.endpoints.results,
|
|
2735
|
+
headers=self._capabilities.auth_headers(),
|
|
2736
|
+
json_body=receipt,
|
|
2737
|
+
)
|
|
2738
|
+
|
|
2739
|
+
try:
|
|
2740
|
+
response, error = _perform_with_retry(
|
|
2741
|
+
perform,
|
|
2742
|
+
retry_policy=self._retry_policy,
|
|
2743
|
+
sleep=self._sleep,
|
|
2744
|
+
rng=self._rng,
|
|
2745
|
+
deadline=deadline,
|
|
2746
|
+
)
|
|
2747
|
+
except (HostedFencedError, HostedChannelFailedError) as exc:
|
|
2748
|
+
self._channel_state.latch(exc)
|
|
2749
|
+
raise
|
|
2750
|
+
if error is not None and error.code == "attempt_superseded": # N22
|
|
2751
|
+
self._channel_state.latch(HostedAttemptSupersededError(error))
|
|
2752
|
+
if error is not None or response is None:
|
|
2753
|
+
return ReceiptPushResult(
|
|
2754
|
+
delivered=False, already_existed=False, error=error
|
|
2755
|
+
)
|
|
2756
|
+
# N20: `200` is read as "already exists / duplicate" per the contract's idempotency rule;
|
|
2757
|
+
# the contract never states the success code for a genuinely NEW receipt (this module's own
|
|
2758
|
+
# `FakePlatform` uses `201`, unconfirmed against the real platform -- see the review report).
|
|
2759
|
+
return ReceiptPushResult(
|
|
2760
|
+
delivered=True, already_existed=response.status_code == 200, error=None
|
|
2761
|
+
)
|
|
2762
|
+
|
|
2763
|
+
|
|
2764
|
+
# =================================================================================================
|
|
2765
|
+
# Channel 3 -- Artifacts. Content-addressed upload + `ArtifactManifestDraft` + delivery.
|
|
2766
|
+
# =================================================================================================
|
|
2767
|
+
|
|
2768
|
+
|
|
2769
|
+
class ArtifactKind(str, Enum):
|
|
2770
|
+
RECORDING_COMBINED = "recording_combined"
|
|
2771
|
+
RECORDING_STEREO = "recording_stereo"
|
|
2772
|
+
RECORDING_CUSTOMER = "recording_customer"
|
|
2773
|
+
RECORDING_ASSISTANT = "recording_assistant"
|
|
2774
|
+
TRANSCRIPT = "transcript"
|
|
2775
|
+
TOOL_TRACE = "tool_trace"
|
|
2776
|
+
RESULT = "result"
|
|
2777
|
+
BUILD = "build"
|
|
2778
|
+
TRACE = "trace"
|
|
2779
|
+
LOG = "log"
|
|
2780
|
+
OTHER = "other"
|
|
2781
|
+
|
|
2782
|
+
|
|
2783
|
+
_RESERVED_ARTIFACT_KINDS = frozenset(
|
|
2784
|
+
{
|
|
2785
|
+
ArtifactKind.BUILD,
|
|
2786
|
+
ArtifactKind.TRANSCRIPT,
|
|
2787
|
+
ArtifactKind.TOOL_TRACE,
|
|
2788
|
+
ArtifactKind.RESULT,
|
|
2789
|
+
}
|
|
2790
|
+
)
|
|
2791
|
+
|
|
2792
|
+
|
|
2793
|
+
def is_reserved_artifact_kind(kind: ArtifactKind) -> bool:
|
|
2794
|
+
""" "the budget is partitioned by reservation: `build` + `transcript` + `tool_trace` + `result`
|
|
2795
|
+
are reserved (always admitted); recordings next; `trace`/`log`/`other` last.\""""
|
|
2796
|
+
return kind in _RESERVED_ARTIFACT_KINDS
|
|
2797
|
+
|
|
2798
|
+
|
|
2799
|
+
_RECORDING_ARTIFACT_KINDS = frozenset(
|
|
2800
|
+
{
|
|
2801
|
+
ArtifactKind.RECORDING_COMBINED,
|
|
2802
|
+
ArtifactKind.RECORDING_STEREO,
|
|
2803
|
+
ArtifactKind.RECORDING_CUSTOMER,
|
|
2804
|
+
ArtifactKind.RECORDING_ASSISTANT,
|
|
2805
|
+
}
|
|
2806
|
+
)
|
|
2807
|
+
|
|
2808
|
+
|
|
2809
|
+
def priority_class(kind: ArtifactKind) -> int:
|
|
2810
|
+
"""N16: the contract's three-tier budget partition as a total order, lower = admitted first --
|
|
2811
|
+
`0` reserved (`is_reserved_artifact_kind`, always admitted), `1` recordings, `2` `trace`/`log`/
|
|
2812
|
+
`other` (admitted last)."""
|
|
2813
|
+
if is_reserved_artifact_kind(kind):
|
|
2814
|
+
return 0
|
|
2815
|
+
if kind in _RECORDING_ARTIFACT_KINDS:
|
|
2816
|
+
return 1
|
|
2817
|
+
return 2
|
|
2818
|
+
|
|
2819
|
+
|
|
2820
|
+
class ArtifactBudgetTracker:
|
|
2821
|
+
"""Client-side mirror of "budget = upload admission ... the guest enforces it first": a
|
|
2822
|
+
per-job cumulative cap across attempts, deduplicated by digest. `would_admit` is a pure check a
|
|
2823
|
+
caller makes before calling `ArtifactsClient.upload` for a non-reserved kind; when it returns
|
|
2824
|
+
`False` the upload is skipped (and named in a `log` event -- the caller's job, not this
|
|
2825
|
+
tracker's). This class does not sequence "refused newest-first" itself -- it has no visibility
|
|
2826
|
+
into candidate ordering across scenarios, which only the scheduler (P9) has; it supplies the
|
|
2827
|
+
admission arithmetic that policy is built on.
|
|
2828
|
+
|
|
2829
|
+
`recording_headroom_bytes` (N16, default 0 -- no behavior change unless a caller opts in):
|
|
2830
|
+
bytes of the remaining budget reserved for recordings not yet seen, subtracted from what a
|
|
2831
|
+
`trace`/`log`/`other` (priority class 2) candidate is allowed to consume. This tracker has no
|
|
2832
|
+
visibility into how many recording bytes are still coming (only the scheduler does), so it
|
|
2833
|
+
cannot give a perfect answer -- reserving a caller-supplied headroom is the honest, testable
|
|
2834
|
+
subset of "recordings next; trace/log/other last" this class alone can enforce.
|
|
2835
|
+
"""
|
|
2836
|
+
|
|
2837
|
+
def __init__(
|
|
2838
|
+
self, max_artifact_bytes: int, *, recording_headroom_bytes: int = 0
|
|
2839
|
+
) -> None:
|
|
2840
|
+
self._max_bytes = max_artifact_bytes
|
|
2841
|
+
self._admitted_bytes = 0
|
|
2842
|
+
self._seen_digests: set[str] = set()
|
|
2843
|
+
self._recording_headroom_bytes = recording_headroom_bytes
|
|
2844
|
+
|
|
2845
|
+
def would_admit(self, kind: ArtifactKind, size: int, *, digest: str) -> bool:
|
|
2846
|
+
if digest in self._seen_digests:
|
|
2847
|
+
return True # already counted; a duplicate upload never grows the budget further
|
|
2848
|
+
tier = priority_class(kind)
|
|
2849
|
+
if tier == 0:
|
|
2850
|
+
return True
|
|
2851
|
+
remaining = self._max_bytes - self._admitted_bytes
|
|
2852
|
+
if tier == 2:
|
|
2853
|
+
remaining -= self._recording_headroom_bytes
|
|
2854
|
+
return size <= remaining
|
|
2855
|
+
|
|
2856
|
+
def record(self, kind: ArtifactKind, size: int, *, digest: str) -> None:
|
|
2857
|
+
del kind # reservation already resolved by would_admit; recorded uniformly here
|
|
2858
|
+
if digest in self._seen_digests:
|
|
2859
|
+
return
|
|
2860
|
+
self._seen_digests.add(digest)
|
|
2861
|
+
self._admitted_bytes += size
|
|
2862
|
+
|
|
2863
|
+
@property
|
|
2864
|
+
def admitted_bytes(self) -> int:
|
|
2865
|
+
return self._admitted_bytes
|
|
2866
|
+
|
|
2867
|
+
|
|
2868
|
+
class ArtifactManifestEntry(BaseModel):
|
|
2869
|
+
model_config = ConfigDict(extra="forbid")
|
|
2870
|
+
|
|
2871
|
+
artifact_id: str
|
|
2872
|
+
kind: ArtifactKind
|
|
2873
|
+
size: int = Field(ge=0)
|
|
2874
|
+
scenario_key: str | None = None
|
|
2875
|
+
|
|
2876
|
+
@model_validator(mode="after")
|
|
2877
|
+
def _validate(self) -> "ArtifactManifestEntry":
|
|
2878
|
+
if not is_valid_digest(self.artifact_id):
|
|
2879
|
+
raise ValueError(f"artifact_id_invalid: {self.artifact_id!r}")
|
|
2880
|
+
return self
|
|
2881
|
+
|
|
2882
|
+
|
|
2883
|
+
class ArtifactManifestDraft(BaseModel):
|
|
2884
|
+
model_config = ConfigDict(extra="forbid")
|
|
2885
|
+
|
|
2886
|
+
schema_version: str = MANIFEST_SCHEMA_VERSION
|
|
2887
|
+
job_id: str = Field(min_length=1)
|
|
2888
|
+
attempt_id: str = Field(min_length=1)
|
|
2889
|
+
attempt_number: int = Field(ge=1)
|
|
2890
|
+
entries: list[ArtifactManifestEntry]
|
|
2891
|
+
complete: bool
|
|
2892
|
+
digest: str
|
|
2893
|
+
|
|
2894
|
+
@model_validator(mode="after")
|
|
2895
|
+
def _validate(self) -> "ArtifactManifestDraft":
|
|
2896
|
+
if self.schema_version != MANIFEST_SCHEMA_VERSION:
|
|
2897
|
+
raise ValueError(f"manifest_schema_unsupported: {self.schema_version}")
|
|
2898
|
+
if not is_valid_digest(self.digest):
|
|
2899
|
+
raise ValueError(f"manifest_digest_invalid: {self.digest!r}")
|
|
2900
|
+
expected = whole_object_digest(self.model_dump(mode="json", exclude={"digest"}))
|
|
2901
|
+
if self.digest != expected:
|
|
2902
|
+
unset = _unset_default_fields(self)
|
|
2903
|
+
hint = (
|
|
2904
|
+
f" -- fields not explicitly set, filled from defaults: {', '.join(unset)}"
|
|
2905
|
+
if unset
|
|
2906
|
+
else ""
|
|
2907
|
+
)
|
|
2908
|
+
raise ValueError(f"manifest_digest_mismatch{hint}")
|
|
2909
|
+
return self
|
|
2910
|
+
|
|
2911
|
+
|
|
2912
|
+
def build_artifact_manifest(
|
|
2913
|
+
*,
|
|
2914
|
+
job_id: str,
|
|
2915
|
+
attempt_id: str,
|
|
2916
|
+
attempt_number: int,
|
|
2917
|
+
entries: list[dict[str, Any]],
|
|
2918
|
+
complete: bool,
|
|
2919
|
+
) -> dict[str, Any]:
|
|
2920
|
+
"""Same pattern as `build_result_receipt`/`build_event_record`: digest computed on the raw
|
|
2921
|
+
input, then re-verified by the model that consumes it."""
|
|
2922
|
+
core: dict[str, Any] = {
|
|
2923
|
+
"schema_version": MANIFEST_SCHEMA_VERSION,
|
|
2924
|
+
"job_id": job_id,
|
|
2925
|
+
"attempt_id": attempt_id,
|
|
2926
|
+
"attempt_number": attempt_number,
|
|
2927
|
+
"entries": entries,
|
|
2928
|
+
"complete": complete,
|
|
2929
|
+
}
|
|
2930
|
+
digest = whole_object_digest(core)
|
|
2931
|
+
draft = ArtifactManifestDraft.model_validate({**core, "digest": digest})
|
|
2932
|
+
return draft.model_dump(mode="json")
|
|
2933
|
+
|
|
2934
|
+
|
|
2935
|
+
@dataclass(frozen=True)
|
|
2936
|
+
class ArtifactUploadResult:
|
|
2937
|
+
delivered: bool
|
|
2938
|
+
already_existed: bool
|
|
2939
|
+
error: ChannelError | None
|
|
2940
|
+
|
|
2941
|
+
|
|
2942
|
+
@dataclass(frozen=True)
|
|
2943
|
+
class ManifestPushResult:
|
|
2944
|
+
delivered: bool
|
|
2945
|
+
already_existed: bool
|
|
2946
|
+
error: ChannelError | None
|
|
2947
|
+
|
|
2948
|
+
|
|
2949
|
+
_DEFAULT_ARTIFACT_CONTENT_TYPES: dict[ArtifactKind, str] = {
|
|
2950
|
+
ArtifactKind.RECORDING_COMBINED: "video/mp4",
|
|
2951
|
+
ArtifactKind.RECORDING_STEREO: "video/mp4",
|
|
2952
|
+
ArtifactKind.RECORDING_CUSTOMER: "video/mp4",
|
|
2953
|
+
ArtifactKind.RECORDING_ASSISTANT: "video/mp4",
|
|
2954
|
+
ArtifactKind.TRANSCRIPT: "application/json",
|
|
2955
|
+
ArtifactKind.RESULT: "application/json",
|
|
2956
|
+
}
|
|
2957
|
+
|
|
2958
|
+
|
|
2959
|
+
def _default_content_type(kind: ArtifactKind) -> str:
|
|
2960
|
+
"""N17: §3a pins recordings to mp4 and `transcript` to a JSON array; a platform serving these
|
|
2961
|
+
back to a UI needs an accurate `Content-Type`, not a blanket octet-stream."""
|
|
2962
|
+
return _DEFAULT_ARTIFACT_CONTENT_TYPES.get(kind, "application/octet-stream")
|
|
2963
|
+
|
|
2964
|
+
|
|
2965
|
+
def _artifact_content_type(kind: ArtifactKind, data: bytes) -> str:
|
|
2966
|
+
"""Return the wire MIME type, preferring the bytes over the nominal format.
|
|
2967
|
+
|
|
2968
|
+
Hosted voice engines currently materialize RIFF/WAVE recordings even though
|
|
2969
|
+
the v1.4 artifact contract's preferred recording format is MP4. Advertising
|
|
2970
|
+
those bytes as ``video/mp4`` makes browsers reject an otherwise valid audio
|
|
2971
|
+
artifact. Keep the contractual default for opaque/test payloads, but sniff
|
|
2972
|
+
the two recording formats we actually support before uploading.
|
|
2973
|
+
"""
|
|
2974
|
+
if kind in {
|
|
2975
|
+
ArtifactKind.RECORDING_COMBINED,
|
|
2976
|
+
ArtifactKind.RECORDING_STEREO,
|
|
2977
|
+
ArtifactKind.RECORDING_CUSTOMER,
|
|
2978
|
+
ArtifactKind.RECORDING_ASSISTANT,
|
|
2979
|
+
}:
|
|
2980
|
+
if len(data) >= 12 and data[:4] == b"RIFF" and data[8:12] == b"WAVE":
|
|
2981
|
+
return "audio/wav"
|
|
2982
|
+
if len(data) >= 12 and data[4:8] == b"ftyp":
|
|
2983
|
+
return "video/mp4"
|
|
2984
|
+
return _default_content_type(kind)
|
|
2985
|
+
|
|
2986
|
+
|
|
2987
|
+
class ArtifactsClient:
|
|
2988
|
+
"""Content-addressed upload (§3a) + manifest delivery (§3b) to `endpoints.artifacts`.
|
|
2989
|
+
|
|
2990
|
+
`upload` verifies the given bytes actually hash to the claimed `artifact_id` BEFORE ever
|
|
2991
|
+
calling the transport -- a local, zero-cost check that catches a caller bug (wrong id, wrong
|
|
2992
|
+
bytes) without spending a round trip on it; the platform's own `422 digest_mismatch` remains
|
|
2993
|
+
the authority for anything this local check cannot see (partial reads, transport corruption).
|
|
2994
|
+
On a `422 digest_mismatch` FROM THE PLATFORM specifically, the contract grants exactly one
|
|
2995
|
+
extra whole-upload retry ("re-upload once, then the referencing scenario is `errored`") --
|
|
2996
|
+
distinct from `_perform_with_retry`'s own loop, which treats every 422 as `PERMANENT_ITEM` and
|
|
2997
|
+
never retries it; this method wraps that loop in one more, narrower retry layer that fires only
|
|
2998
|
+
for that one code.
|
|
2999
|
+
|
|
3000
|
+
Size accounting: `X-Artifact-Size` is never a caller-supplied value -- it is always derived
|
|
3001
|
+
from `len(data)`, the same bytes actually transmitted, so a `422 size_mismatch` against what
|
|
3002
|
+
this client sends is structurally unreachable from here (the platform's own count remains the
|
|
3003
|
+
authority for what actually arrived over the wire).
|
|
3004
|
+
|
|
3005
|
+
N18: once a 413 `artifact_budget_exceeded` is observed, this instance latches locally -- every
|
|
3006
|
+
later `upload()` for a NON-reserved kind is refused without contacting the platform at all
|
|
3007
|
+
("stop uploading non-reserved kinds, log, continue the run"); reserved kinds keep uploading
|
|
3008
|
+
(they are always admitted, budget or not).
|
|
3009
|
+
"""
|
|
3010
|
+
|
|
3011
|
+
def __init__(
|
|
3012
|
+
self,
|
|
3013
|
+
capabilities: HostedCapabilities,
|
|
3014
|
+
transport: Transport | None = None,
|
|
3015
|
+
*,
|
|
3016
|
+
retry_policy: RetryPolicy | None = None,
|
|
3017
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
3018
|
+
rng: Callable[[], float] = random.random,
|
|
3019
|
+
chunk_threshold_bytes: int = ARTIFACT_CHUNKED_UPLOAD_THRESHOLD_BYTES,
|
|
3020
|
+
chunk_size_bytes: int = 8 * 1024 * 1024,
|
|
3021
|
+
channel_state: ChannelState | None = None,
|
|
3022
|
+
) -> None:
|
|
3023
|
+
self._capabilities = capabilities
|
|
3024
|
+
self._transport = transport or RequestsTransport()
|
|
3025
|
+
self._retry_policy = retry_policy or RetryPolicy()
|
|
3026
|
+
self._sleep = sleep
|
|
3027
|
+
self._rng = rng
|
|
3028
|
+
self._chunk_threshold_bytes = chunk_threshold_bytes
|
|
3029
|
+
self._chunk_size_bytes = chunk_size_bytes
|
|
3030
|
+
self._channel_state = channel_state or ChannelState()
|
|
3031
|
+
self._budget_exhausted = False # N18
|
|
3032
|
+
|
|
3033
|
+
def upload(
|
|
3034
|
+
self,
|
|
3035
|
+
artifact_id_hex: str,
|
|
3036
|
+
data: bytes,
|
|
3037
|
+
*,
|
|
3038
|
+
kind: ArtifactKind,
|
|
3039
|
+
scenario_key: str | None = None,
|
|
3040
|
+
content_type: str | None = None,
|
|
3041
|
+
deadline: float | None = None,
|
|
3042
|
+
) -> ArtifactUploadResult:
|
|
3043
|
+
self._channel_state.check()
|
|
3044
|
+
if self._budget_exhausted and not is_reserved_artifact_kind(kind):
|
|
3045
|
+
logger.warning(
|
|
3046
|
+
"artifacts upload: budget already exhausted (413 artifact_budget_exceeded observed "
|
|
3047
|
+
"earlier this attempt) -- skipping non-reserved kind=%s without contacting the "
|
|
3048
|
+
"platform",
|
|
3049
|
+
kind.value,
|
|
3050
|
+
)
|
|
3051
|
+
return ArtifactUploadResult(
|
|
3052
|
+
delivered=False,
|
|
3053
|
+
already_existed=False,
|
|
3054
|
+
error=ChannelError(
|
|
3055
|
+
ChannelOutcome.BUDGET_EXCEEDED,
|
|
3056
|
+
None,
|
|
3057
|
+
"artifact_budget_exceeded",
|
|
3058
|
+
"budget already exhausted for this attempt (latched locally)",
|
|
3059
|
+
status_code=None,
|
|
3060
|
+
),
|
|
3061
|
+
)
|
|
3062
|
+
if not re.fullmatch(r"[0-9a-f]{64}", artifact_id_hex):
|
|
3063
|
+
raise ValueError(f"artifact_id_invalid: {artifact_id_hex!r}")
|
|
3064
|
+
if artifact_id_hex == "manifest":
|
|
3065
|
+
# Unreachable through the hex check above (`manifest` isn't 64 hex chars) but the
|
|
3066
|
+
# contract calls this collision out by name ("`manifest` is a reserved segment").
|
|
3067
|
+
raise ValueError("artifact_id_reserved: manifest")
|
|
3068
|
+
computed = hashlib.sha256(data).hexdigest()
|
|
3069
|
+
if computed != artifact_id_hex:
|
|
3070
|
+
raise ValueError(
|
|
3071
|
+
f"artifact_digest_mismatch_local: expected {artifact_id_hex}, computed {computed}"
|
|
3072
|
+
)
|
|
3073
|
+
|
|
3074
|
+
url = f"{self._capabilities.endpoints.artifacts}{artifact_id_hex}/"
|
|
3075
|
+
size = len(data)
|
|
3076
|
+
headers = {
|
|
3077
|
+
**self._capabilities.auth_headers(),
|
|
3078
|
+
"X-Artifact-Kind": kind.value,
|
|
3079
|
+
"X-Artifact-Size": str(size),
|
|
3080
|
+
"Content-Type": content_type or _artifact_content_type(kind, data),
|
|
3081
|
+
}
|
|
3082
|
+
if scenario_key is not None:
|
|
3083
|
+
headers["X-Scenario-Key"] = scenario_key
|
|
3084
|
+
|
|
3085
|
+
def perform(_attempt: int) -> TransportResponse:
|
|
3086
|
+
# A fresh generator per attempt when chunked -- see `_iter_chunks`.
|
|
3087
|
+
body: bytes | Iterator[bytes] = (
|
|
3088
|
+
_iter_chunks(data, self._chunk_size_bytes)
|
|
3089
|
+
if size > self._chunk_threshold_bytes
|
|
3090
|
+
else data
|
|
3091
|
+
)
|
|
3092
|
+
return self._transport.request("PUT", url, headers=headers, data=body)
|
|
3093
|
+
|
|
3094
|
+
response: TransportResponse | None = None
|
|
3095
|
+
error: ChannelError | None = None
|
|
3096
|
+
for outer_attempt in range(
|
|
3097
|
+
2
|
|
3098
|
+
): # "re-upload once" on a platform-confirmed digest mismatch
|
|
3099
|
+
try:
|
|
3100
|
+
response, error = _perform_with_retry(
|
|
3101
|
+
perform,
|
|
3102
|
+
retry_policy=self._retry_policy,
|
|
3103
|
+
sleep=self._sleep,
|
|
3104
|
+
rng=self._rng,
|
|
3105
|
+
deadline=deadline,
|
|
3106
|
+
)
|
|
3107
|
+
except (HostedFencedError, HostedChannelFailedError) as exc:
|
|
3108
|
+
self._channel_state.latch(exc)
|
|
3109
|
+
raise
|
|
3110
|
+
if not (
|
|
3111
|
+
error is not None
|
|
3112
|
+
and error.code == "digest_mismatch"
|
|
3113
|
+
and outer_attempt == 0
|
|
3114
|
+
):
|
|
3115
|
+
break
|
|
3116
|
+
|
|
3117
|
+
if error is not None and error.outcome is ChannelOutcome.BUDGET_EXCEEDED:
|
|
3118
|
+
self._budget_exhausted = True # N18
|
|
3119
|
+
|
|
3120
|
+
if error is not None or response is None:
|
|
3121
|
+
return ArtifactUploadResult(
|
|
3122
|
+
delivered=False, already_existed=False, error=error
|
|
3123
|
+
)
|
|
3124
|
+
# N20: `200` is read as "already exists" per the contract's content-addressed upload
|
|
3125
|
+
# semantics; the success code for a genuinely NEW upload is `201` (this module's own
|
|
3126
|
+
# `FakePlatform` matches that but it is unconfirmed against the real platform).
|
|
3127
|
+
return ArtifactUploadResult(
|
|
3128
|
+
delivered=True, already_existed=response.status_code == 200, error=None
|
|
3129
|
+
)
|
|
3130
|
+
|
|
3131
|
+
def push_manifest(
|
|
3132
|
+
self, manifest: dict[str, Any], *, deadline: float | None = None
|
|
3133
|
+
) -> ManifestPushResult:
|
|
3134
|
+
self._channel_state.check()
|
|
3135
|
+
url = f"{self._capabilities.endpoints.artifacts}manifest/"
|
|
3136
|
+
|
|
3137
|
+
def perform(_attempt: int) -> TransportResponse:
|
|
3138
|
+
return self._transport.request(
|
|
3139
|
+
"POST",
|
|
3140
|
+
url,
|
|
3141
|
+
headers=self._capabilities.auth_headers(),
|
|
3142
|
+
json_body=manifest,
|
|
3143
|
+
)
|
|
3144
|
+
|
|
3145
|
+
try:
|
|
3146
|
+
response, error = _perform_with_retry(
|
|
3147
|
+
perform,
|
|
3148
|
+
retry_policy=self._retry_policy,
|
|
3149
|
+
sleep=self._sleep,
|
|
3150
|
+
rng=self._rng,
|
|
3151
|
+
deadline=deadline,
|
|
3152
|
+
)
|
|
3153
|
+
except (HostedFencedError, HostedChannelFailedError) as exc:
|
|
3154
|
+
self._channel_state.latch(exc)
|
|
3155
|
+
raise
|
|
3156
|
+
if error is not None and error.code == "attempt_superseded": # N22
|
|
3157
|
+
self._channel_state.latch(HostedAttemptSupersededError(error))
|
|
3158
|
+
if error is not None or response is None:
|
|
3159
|
+
return ManifestPushResult(
|
|
3160
|
+
delivered=False, already_existed=False, error=error
|
|
3161
|
+
)
|
|
3162
|
+
# N20: same caveat as receipts/uploads above -- `200` == duplicate is contract-stated,
|
|
3163
|
+
# the new-manifest success code is `201` per `FakePlatform`, unconfirmed against the real
|
|
3164
|
+
# platform.
|
|
3165
|
+
return ManifestPushResult(
|
|
3166
|
+
delivered=True, already_existed=response.status_code == 200, error=None
|
|
3167
|
+
)
|
|
3168
|
+
|
|
3169
|
+
|
|
3170
|
+
__all__ = [
|
|
3171
|
+
"ARTIFACT_CHUNKED_UPLOAD_THRESHOLD_BYTES",
|
|
3172
|
+
"CAPABILITIES_PATH",
|
|
3173
|
+
"CAPABILITIES_SCHEMA_VERSION",
|
|
3174
|
+
"EVENTS_MAX_BATCH",
|
|
3175
|
+
"EVENTS_MAX_BATCH_BYTES",
|
|
3176
|
+
"EVENT_PAYLOAD_MAX_BYTES",
|
|
3177
|
+
"EVENT_SCHEMA_VERSION",
|
|
3178
|
+
"FLUSH_WINDOW_SECONDS",
|
|
3179
|
+
"MANIFEST_SCHEMA_VERSION",
|
|
3180
|
+
"RESULT_SCHEMA_VERSION",
|
|
3181
|
+
"ArtifactBudgetTracker",
|
|
3182
|
+
"ArtifactKind",
|
|
3183
|
+
"ArtifactManifestDraft",
|
|
3184
|
+
"ArtifactManifestEntry",
|
|
3185
|
+
"ArtifactUploadResult",
|
|
3186
|
+
"ArtifactsClient",
|
|
3187
|
+
"BaselineFrozenPayload",
|
|
3188
|
+
"BaselineInputsChangedPayload",
|
|
3189
|
+
"CallSummary",
|
|
3190
|
+
"CapabilitiesError",
|
|
3191
|
+
"ChannelError",
|
|
3192
|
+
"ChannelOutcome",
|
|
3193
|
+
"ChannelState",
|
|
3194
|
+
"CheckpointEvaluation",
|
|
3195
|
+
"DegradeReason",
|
|
3196
|
+
"EvaluationResult",
|
|
3197
|
+
"EventsClient",
|
|
3198
|
+
"EventsFlushResult",
|
|
3199
|
+
"HostedAttemptSupersededError",
|
|
3200
|
+
"HostedCapabilities",
|
|
3201
|
+
"HostedChannelFailedError",
|
|
3202
|
+
"HostedEndpoints",
|
|
3203
|
+
"HostedEvent",
|
|
3204
|
+
"HostedEventDraft",
|
|
3205
|
+
"HostedFencedError",
|
|
3206
|
+
"LogLevel",
|
|
3207
|
+
"LogPayload",
|
|
3208
|
+
"ManifestPushResult",
|
|
3209
|
+
"MetricEvaluation",
|
|
3210
|
+
"OutboundError",
|
|
3211
|
+
"OutboundEventType",
|
|
3212
|
+
"OutboundSpool",
|
|
3213
|
+
"OutboundSpoolError",
|
|
3214
|
+
"ParallelismDegradedPayload",
|
|
3215
|
+
"ReceiptPushResult",
|
|
3216
|
+
"RequestsTransport",
|
|
3217
|
+
"ResultReceiptDraft",
|
|
3218
|
+
"ResultsClient",
|
|
3219
|
+
"RetryPolicy",
|
|
3220
|
+
"ScenarioCounts",
|
|
3221
|
+
"ScenarioRetriedPayload",
|
|
3222
|
+
"ScenarioStartedPayload",
|
|
3223
|
+
"ScenarioStatus",
|
|
3224
|
+
"SpooledRecord",
|
|
3225
|
+
"StageChangedPayload",
|
|
3226
|
+
"SubGoalResult",
|
|
3227
|
+
"TerminalFailure",
|
|
3228
|
+
"TerminalPayload",
|
|
3229
|
+
"TerminalReason",
|
|
3230
|
+
"Transport",
|
|
3231
|
+
"TransportError",
|
|
3232
|
+
"TransportResponse",
|
|
3233
|
+
"WorldUnhealthyPayload",
|
|
3234
|
+
"build_artifact_manifest",
|
|
3235
|
+
"build_event_record",
|
|
3236
|
+
"build_result_receipt",
|
|
3237
|
+
"build_skipped_receipt",
|
|
3238
|
+
"canonical_bytes",
|
|
3239
|
+
"classify_response",
|
|
3240
|
+
"compute_backoff_seconds",
|
|
3241
|
+
"event_payload_digest",
|
|
3242
|
+
"format_rfc3339_millis",
|
|
3243
|
+
"is_reserved_artifact_kind",
|
|
3244
|
+
"is_valid_digest",
|
|
3245
|
+
"is_valid_rfc3339_millis",
|
|
3246
|
+
"load_capabilities",
|
|
3247
|
+
"priority_class",
|
|
3248
|
+
"redact_outbound_text",
|
|
3249
|
+
"sha256_digest",
|
|
3250
|
+
"truncate_log_message",
|
|
3251
|
+
"whole_object_digest",
|
|
3252
|
+
]
|