agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,2402 @@
|
|
|
1
|
+
"""The hosted guest's `main()` — `hosted-execution-seams.md` v1.14 §0/§4/§5, `outbound-channels.md`
|
|
2
|
+
v1.3, `world-handle-interface.md` v3.4. Everything between "sandbox starts" and "exit code": read
|
|
3
|
+
`/work/job.json`, load the platform capability file, run §2e preflight, pre-allocate scenarios
|
|
4
|
+
against `endpoints.scenarios`, provision the world pool, drive the scenario loop, adapt its events/
|
|
5
|
+
receipts/artifacts onto the real outbound clients, and honor the exit-code contract (§0.6).
|
|
6
|
+
|
|
7
|
+
Ownership boundary (read this before touching orchestration order): the stages BEFORE bundle
|
|
8
|
+
authoring — `understanding_agent`, `generating_environment`, `building_environment`'s bundle-write
|
|
9
|
+
half — belong to Rishav's stages (contract §6) and are not implemented anywhere in this repo yet.
|
|
10
|
+
This module does not attempt them. `BundleSource`/`ScenarioSource` below are the seams a later
|
|
11
|
+
change wires the real stages through; until then their defaults raise a typed, clearly-named error
|
|
12
|
+
rather than silently producing a fake bundle or a fake scenario set.
|
|
13
|
+
|
|
14
|
+
`process_runtime.py`, `hosted_scheduler.py`, and `outbound.py` were being fixed by parallel workers
|
|
15
|
+
while this module was written. It codes against the four frozen contracts and the cross-review
|
|
16
|
+
obligation lists, not against those files' exact HEAD.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import asyncio
|
|
23
|
+
import hashlib
|
|
24
|
+
import json
|
|
25
|
+
import logging
|
|
26
|
+
import os
|
|
27
|
+
import random
|
|
28
|
+
import signal
|
|
29
|
+
import time
|
|
30
|
+
import uuid
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from datetime import datetime, timezone
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Any, Callable, Protocol, Sequence
|
|
35
|
+
|
|
36
|
+
from . import observability
|
|
37
|
+
from . import outbound as ob
|
|
38
|
+
from .bundle_v2 import BundleV2Error, EnvironmentBundleV2, load_bundle_v2
|
|
39
|
+
from .call_runner import CallRunnerContext, CallRunnerImpl
|
|
40
|
+
from .hosted_scheduler import (
|
|
41
|
+
CallOutcome,
|
|
42
|
+
CallRunner,
|
|
43
|
+
HostedScheduler,
|
|
44
|
+
ResultReceipt,
|
|
45
|
+
RunResult,
|
|
46
|
+
Scenario,
|
|
47
|
+
World,
|
|
48
|
+
WorldFactory,
|
|
49
|
+
WorldPool,
|
|
50
|
+
WorldProvisioner,
|
|
51
|
+
)
|
|
52
|
+
from .job import (
|
|
53
|
+
ArtifactLevel,
|
|
54
|
+
ExecutionMode,
|
|
55
|
+
FailureDomain,
|
|
56
|
+
HarnessArtifactPolicy,
|
|
57
|
+
HarnessJob,
|
|
58
|
+
HarnessStage,
|
|
59
|
+
)
|
|
60
|
+
from .process_preflight import PreflightError, preflight_bundle
|
|
61
|
+
from .process_runtime import (
|
|
62
|
+
SECTION_2F_DOMAIN,
|
|
63
|
+
EnvironmentRuntime,
|
|
64
|
+
ProcessRuntimeError,
|
|
65
|
+
ProcessRuntimeProvider,
|
|
66
|
+
RuntimeEndpoint,
|
|
67
|
+
)
|
|
68
|
+
from .scenario_source import (
|
|
69
|
+
BundleScenarioSource,
|
|
70
|
+
ScenarioDocumentInvalid,
|
|
71
|
+
bundle_has_scenarios,
|
|
72
|
+
)
|
|
73
|
+
from .world.handle import HostedWorld
|
|
74
|
+
from .world.stores.postgres import AttachedPostgresStore
|
|
75
|
+
|
|
76
|
+
logger = logging.getLogger(__name__)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class _JobIdFilter(logging.Filter):
|
|
80
|
+
"""Stamp every log record with the job this runner is serving.
|
|
81
|
+
|
|
82
|
+
One runner process serves exactly one job, so the id is process-wide rather than per-task
|
|
83
|
+
state. Concurrent runs are separate processes, but their stdout is collected into one place,
|
|
84
|
+
and a line with no job id cannot be attributed to a run at all -- which is the difference
|
|
85
|
+
between reading a log and guessing at it.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
def __init__(self) -> None:
|
|
89
|
+
super().__init__()
|
|
90
|
+
self.job_id = "-"
|
|
91
|
+
|
|
92
|
+
def filter(self, record: logging.LogRecord) -> bool:
|
|
93
|
+
if not hasattr(record, "job_id"):
|
|
94
|
+
record.job_id = self.job_id
|
|
95
|
+
return True
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
_JOB_ID_FILTER = _JobIdFilter()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def configure_runner_logging(job_id: str | None) -> None:
|
|
102
|
+
"""Put the job id on every line this process emits, including libraries' lines.
|
|
103
|
+
|
|
104
|
+
Installed on the root logger rather than ours, because the lines that are hardest to attribute
|
|
105
|
+
are the ones from livekit, httpx and the model clients.
|
|
106
|
+
"""
|
|
107
|
+
_JOB_ID_FILTER.job_id = str(job_id or "-")
|
|
108
|
+
root = logging.getLogger()
|
|
109
|
+
for handler in root.handlers:
|
|
110
|
+
handler.addFilter(_JOB_ID_FILTER)
|
|
111
|
+
if not root.handlers:
|
|
112
|
+
handler = logging.StreamHandler()
|
|
113
|
+
handler.addFilter(_JOB_ID_FILTER)
|
|
114
|
+
root.addHandler(handler)
|
|
115
|
+
root.setLevel(logging.INFO)
|
|
116
|
+
for handler in logging.getLogger().handlers:
|
|
117
|
+
handler.setFormatter(
|
|
118
|
+
logging.Formatter(
|
|
119
|
+
"%(asctime)s %(levelname)s job=%(job_id)s %(name)s: %(message)s"
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
# --- §0.6 exit-code contract --------------------------------------------------------------------
|
|
124
|
+
#
|
|
125
|
+
# 0 = any terminal stage reached (completed/failed/canceled), outbox flushed. 3 = fenced/superseded
|
|
126
|
+
# (HostedFencedError anywhere -> stop emitting, no terminal event, exit 3). 4 = the terminal was
|
|
127
|
+
# decided but the final drain could not deliver it (the events channel failed, or the platform
|
|
128
|
+
# permanently rejected the terminal item itself) -- the gateway treats it exactly like a crash
|
|
129
|
+
# (infrastructure retry, fresh channels), but the distinct code tells operators the job DID reach a
|
|
130
|
+
# terminal state, unlike a genuine crash. Any other non-zero = the guest crashed before a terminal
|
|
131
|
+
# state -- the gateway records `infrastructure`. Capabilities-file failures are explicitly carved
|
|
132
|
+
# out of the "any other non-zero" bucket only by CODE (they must never be 3, per
|
|
133
|
+
# outbound-channels.md v1.3's rejection table); they still use a non-zero exit here since there is
|
|
134
|
+
# no channel to report a terminal FAILED event through.
|
|
135
|
+
EXIT_OK = 0
|
|
136
|
+
EXIT_FENCED = 3
|
|
137
|
+
EXIT_TERMINAL_UNDELIVERED = (
|
|
138
|
+
4 # terminal reached but not provably flushed on the final drain.
|
|
139
|
+
)
|
|
140
|
+
EXIT_BOOT_FAILURE = (
|
|
141
|
+
1 # capabilities.json could not be loaded -- no channel, no event (v1.3 table).
|
|
142
|
+
)
|
|
143
|
+
EXIT_CRASHED = 2 # an uncaught failure before any terminal stage was reached.
|
|
144
|
+
|
|
145
|
+
# Cancellation signal (spine §0 step 7 / outbound-channels.md "Cancellation signal"). The task
|
|
146
|
+
# brief that spawned this module named `/work/cancel.json`; the two frozen contracts that actually
|
|
147
|
+
# define this file (seams §0 step 7, outbound-channels "Cancellation signal") both name
|
|
148
|
+
# `/run/futureagi/cancel.json`. Contracts are authoritative over a task brief.
|
|
149
|
+
CANCEL_SIGNAL_PATH = "/run/futureagi/cancel.json"
|
|
150
|
+
|
|
151
|
+
# STUCK DECISION (fail-safe/reversible; contract gap): the invocation contract
|
|
152
|
+
# (spine §0 step 5) pins the entrypoint's argv to exactly `job --source ... --output ...`; `--output`
|
|
153
|
+
# is `/work/artifacts` (spine layout block), so `work_directory` (what `preflight_bundle`/
|
|
154
|
+
# `provision`/`write_build_output` all want -- the `/work` root) is derived as `output.parent`
|
|
155
|
+
# rather than taken as a separate flag, since the frozen invocation line has no room for one.
|
|
156
|
+
# `bundle_dir` has no convention anywhere in the frozen documents at all (bundle authoring is not
|
|
157
|
+
# built yet); `DEFAULT_BUNDLE_DIR_NAME` is this module's own placeholder location, overridable via
|
|
158
|
+
# `BundleSource` injection so a later change can point it at wherever the real authoring stage ends
|
|
159
|
+
# up writing without touching this file's orchestration.
|
|
160
|
+
DEFAULT_BUNDLE_DIR_NAME = "bundle"
|
|
161
|
+
EVENTS_SPOOL_DIR_NAME = (
|
|
162
|
+
"outbound-spool" # must not live under work_directory/"artifacts".
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
SECRETS_PATH = Path("/run/futureagi/secrets.json")
|
|
166
|
+
SIMULATOR_SECRETS_PATH = Path("/run/futureagi/simulator-secrets.json")
|
|
167
|
+
|
|
168
|
+
# Platform-owned simulator configuration is delivered on a separate control-plane channel. It
|
|
169
|
+
# must never be confused with the customer's ``target_provider`` refs, which are selectively
|
|
170
|
+
# injected into the untrusted agent processes by ProcessRuntimeProvider. These names are the
|
|
171
|
+
# complete set the in-process text/voice simulators may consume.
|
|
172
|
+
_SIMULATOR_SECRET_ALIASES = frozenset(
|
|
173
|
+
{
|
|
174
|
+
"ALK_BACKGROUND_NOISE",
|
|
175
|
+
"ALK_BACKGROUND_NOISE_CATALOG",
|
|
176
|
+
"ALK_HARNESS",
|
|
177
|
+
"ALK_VOICEMAIL_SCENARIOS",
|
|
178
|
+
"ALK_HARNESS_MODEL",
|
|
179
|
+
"ALK_HARNESS_THINKING",
|
|
180
|
+
"ALK_VERTEX_LOCATION",
|
|
181
|
+
"CARTESIA_API_KEY",
|
|
182
|
+
"DEEPGRAM_API_KEY",
|
|
183
|
+
# Observe configuration: the platform's own account, never the customer's.
|
|
184
|
+
"FI_API_KEY",
|
|
185
|
+
"FI_BASE_URL",
|
|
186
|
+
"FI_HARNESS_PROJECT",
|
|
187
|
+
"FI_SECRET_KEY",
|
|
188
|
+
"GEMINI_API_KEY",
|
|
189
|
+
"GOOGLE_API_KEY",
|
|
190
|
+
"GOOGLE_APPLICATION_CREDENTIALS",
|
|
191
|
+
"GOOGLE_CLOUD_LOCATION",
|
|
192
|
+
"GOOGLE_CLOUD_PROJECT",
|
|
193
|
+
"GOOGLE_GENAI_USE_VERTEXAI",
|
|
194
|
+
"HARNESS_BACKGROUND_NOISE_VOLUME",
|
|
195
|
+
"HARNESS_OBSERVABILITY",
|
|
196
|
+
"LIVEKIT_URL",
|
|
197
|
+
"LIVEKIT_API_KEY",
|
|
198
|
+
"LIVEKIT_API_SECRET",
|
|
199
|
+
"OPENAI_API_KEY",
|
|
200
|
+
"SIMULATOR_LLM_MODEL",
|
|
201
|
+
"SIMULATOR_LLM_PROVIDER",
|
|
202
|
+
"SIMULATOR_STT_MODEL",
|
|
203
|
+
"SIMULATOR_STT_PROVIDER",
|
|
204
|
+
"SIMULATOR_TTS_MODEL",
|
|
205
|
+
"SIMULATOR_TTS_PROVIDER",
|
|
206
|
+
}
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
# =================================================================================================
|
|
211
|
+
# Boot -- job.json + capabilities.json (§0.2/§0.4; outbound-channels.md Authentication).
|
|
212
|
+
# =================================================================================================
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def load_job(job_path: Path) -> HarnessJob:
|
|
216
|
+
"""§0.2: `/work/job.json` is the provisioner's job-identity and configuration source."""
|
|
217
|
+
job = HarnessJob.model_validate_json(job_path.read_text(encoding="utf-8"))
|
|
218
|
+
if job.execution is not ExecutionMode.HOSTED:
|
|
219
|
+
raise ValueError("hosted_entrypoint_requires_hosted_job")
|
|
220
|
+
return job
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def resolve_parallelism(job: HarnessJob) -> int:
|
|
224
|
+
"""`job.runtime.parallelism` = W (glossary). Returns the RAW requested value, never
|
|
225
|
+
clamped -- §2e.7 reserves `parallelism_out_of_range` for a W outside 1..8, and
|
|
226
|
+
`preflight_bundle` (called BEFORE any provisioning) is the enforcement point for the UPPER
|
|
227
|
+
bound. The lower bound never reaches preflight at all: `RuntimeRequirements.parallelism`'s own
|
|
228
|
+
`ge=1` rejects a non-positive W earlier, at `load_job`, as a deliberate defense-in-depth floor
|
|
229
|
+
(harmless today since the gateway caps W at admission before a job is ever built). Clamping
|
|
230
|
+
here would silently launder an in-range-but-wrong W and make `parallelism_out_of_range`
|
|
231
|
+
permanently unreachable for the upper bound."""
|
|
232
|
+
return job.runtime.parallelism
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def job_secret_purposes(job: HarnessJob) -> dict[str, str]:
|
|
236
|
+
"""§1: `agent.secret_refs` alias -> `SecretRef.purpose`, the shape `preflight_bundle` wants."""
|
|
237
|
+
return {alias: ref.purpose for alias, ref in job.agent.secret_refs.items()}
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def peek_secret_values(secrets_path: Path) -> tuple[str, ...]:
|
|
241
|
+
"""A non-destructive read of `/run/futureagi/secrets.json`'s VALUES ONLY, for outbound
|
|
242
|
+
redaction (`extra_secret_values` — outbound.py's `redact_outbound_text`). §0.3's lifetime rule
|
|
243
|
+
("the provisioner loads this file into memory at startup and deletes it") is honored by
|
|
244
|
+
`ProcessRuntimeProvider` itself; this is an additional, side-effect-free read (no unlink) done
|
|
245
|
+
once at boot so free-text event/log/failure fields can be scrubbed of every resolved secret
|
|
246
|
+
value, not just URL userinfo. Never fatal: a missing/malformed file just means no extra values
|
|
247
|
+
to scrub, matching `redact_outbound_text`'s own `extra_secret_values=()` default."""
|
|
248
|
+
try:
|
|
249
|
+
raw = json.loads(secrets_path.read_text(encoding="utf-8"))
|
|
250
|
+
except (OSError, ValueError):
|
|
251
|
+
return ()
|
|
252
|
+
if not isinstance(raw, dict):
|
|
253
|
+
return ()
|
|
254
|
+
return tuple(str(value) for value in raw.values() if value)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def peek_secret_values_for_purpose(
|
|
258
|
+
secrets_path: Path,
|
|
259
|
+
secret_purposes: dict[str, str],
|
|
260
|
+
purpose: str,
|
|
261
|
+
) -> dict[str, str]:
|
|
262
|
+
"""The same non-destructive, no-unlink read as `peek_secret_values` (same file, same timing
|
|
263
|
+
constraint -- called BEFORE `pool.start()`, which is what actually deletes the file), but
|
|
264
|
+
ALIAS-preserving and filtered to one explicit purpose -- `peek_secret_values` throws the
|
|
265
|
+
alias away, which is fine for outbound redaction (it only needs the raw values) but useless for
|
|
266
|
+
the real `CallRunner`, which needs to pick e.g. `LIVEKIT_API_KEY` out of the map by name. Never
|
|
267
|
+
fatal: a missing/malformed file just means no target-provider secrets are available yet,
|
|
268
|
+
matching `CallRunnerImpl`'s own pre-dial validation (it reports the gap as a typed
|
|
269
|
+
`CallAborted`, never crashes on an empty map)."""
|
|
270
|
+
try:
|
|
271
|
+
raw = json.loads(secrets_path.read_text(encoding="utf-8"))
|
|
272
|
+
except (OSError, ValueError):
|
|
273
|
+
return {}
|
|
274
|
+
if not isinstance(raw, dict):
|
|
275
|
+
return {}
|
|
276
|
+
return {
|
|
277
|
+
str(alias): str(value)
|
|
278
|
+
for alias, value in raw.items()
|
|
279
|
+
if secret_purposes.get(str(alias)) == purpose
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def peek_target_provider_secret_values(
|
|
284
|
+
secrets_path: Path, secret_purposes: dict[str, str]
|
|
285
|
+
) -> dict[str, str]:
|
|
286
|
+
return peek_secret_values_for_purpose(
|
|
287
|
+
secrets_path, secret_purposes, "target_provider"
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def peek_simulator_provider_secret_values(
|
|
292
|
+
secrets_path: Path, secret_purposes: dict[str, str]
|
|
293
|
+
) -> dict[str, str]:
|
|
294
|
+
return peek_secret_values_for_purpose(
|
|
295
|
+
secrets_path, secret_purposes, "simulator_provider"
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def load_simulator_secret_values(path: Path) -> dict[str, str]:
|
|
300
|
+
"""Load and immediately remove the platform-owned simulator secret channel.
|
|
301
|
+
|
|
302
|
+
The fixed allowlist is deliberate: a platform deployment cannot accidentally use this file
|
|
303
|
+
to inject arbitrary ambient variables into the control process. Agent subprocesses still do
|
|
304
|
+
not inherit these values because ``process_runtime`` starts them from its closed environment
|
|
305
|
+
allowlist plus purpose-matched target secrets.
|
|
306
|
+
"""
|
|
307
|
+
try:
|
|
308
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
309
|
+
except (OSError, ValueError):
|
|
310
|
+
return {}
|
|
311
|
+
finally:
|
|
312
|
+
try:
|
|
313
|
+
path.unlink(missing_ok=True)
|
|
314
|
+
except OSError as exc:
|
|
315
|
+
logger.warning("simulator-secrets.json unlink failed: %s", exc)
|
|
316
|
+
if not isinstance(raw, dict):
|
|
317
|
+
return {}
|
|
318
|
+
return {
|
|
319
|
+
str(alias): str(value)
|
|
320
|
+
for alias, value in raw.items()
|
|
321
|
+
if str(alias) in _SIMULATOR_SECRET_ALIASES and value not in (None, "")
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
# =================================================================================================
|
|
326
|
+
# Bundle source -- §2 bundle authoring is not this module's (or built anywhere yet); injectable.
|
|
327
|
+
# =================================================================================================
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
class BundleUnavailableError(RuntimeError):
|
|
331
|
+
"""Raised by a `BundleSource` when no bundle could be produced/located. Mapped the same way as
|
|
332
|
+
a `PreflightError` (FAILED, `FailureDomain.ENVIRONMENT`, stage `validating_environment`) —
|
|
333
|
+
from the entrypoint's point of view "no bundle" and "bad bundle" are the same class of
|
|
334
|
+
environment-authoring fault, and §2e's own failure table has no separate code for it."""
|
|
335
|
+
|
|
336
|
+
def __init__(self, code: str, message: str) -> None:
|
|
337
|
+
self.code = code
|
|
338
|
+
self.message = message
|
|
339
|
+
super().__init__(f"{code}: {message}")
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
class BundleSource(Protocol):
|
|
343
|
+
def load(
|
|
344
|
+
self, job: HarnessJob, *, source: Path, work_directory: Path
|
|
345
|
+
) -> tuple[EnvironmentBundleV2, Path]: ...
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
# §2e's closed failure-code table (hosted-execution-seams.md) -- `BundleV2Error` has no typed
|
|
349
|
+
# `.code` (a bare `RuntimeError`), so `DefaultBundleSource.load` below string-splits
|
|
350
|
+
# its message on ":". `bundle_manifest_missing` (one of the four messages `load_bundle_v2` can
|
|
351
|
+
# raise) is not in this table -- a real contract gap -- so both that code AND anything else the
|
|
352
|
+
# split produces outside this frozen set fall back to `bundle_manifest_invalid` rather than
|
|
353
|
+
# shipping an unlisted code across the outbound seam.
|
|
354
|
+
_SECTION_2E_CODES = frozenset(
|
|
355
|
+
{
|
|
356
|
+
"compose_not_hosted",
|
|
357
|
+
"engine_unsupported",
|
|
358
|
+
"no_sql_store",
|
|
359
|
+
"seed_missing",
|
|
360
|
+
"seed_strategy_unsupported",
|
|
361
|
+
"sentinel_shape_mismatch",
|
|
362
|
+
"store_protocol_unsupported",
|
|
363
|
+
"capability_engine_mismatch",
|
|
364
|
+
"store_service_not_managed",
|
|
365
|
+
"reserved_name",
|
|
366
|
+
"unknown_placeholder",
|
|
367
|
+
"unknown_field",
|
|
368
|
+
"secret_in_bundle",
|
|
369
|
+
"secret_unclaimed",
|
|
370
|
+
"secret_missing",
|
|
371
|
+
"secret_purpose_forbidden",
|
|
372
|
+
"build_requires_root",
|
|
373
|
+
"user_assignment_invalid",
|
|
374
|
+
"configuration_name_duplicate",
|
|
375
|
+
"configuration_name_required",
|
|
376
|
+
"configuration_name_reserved",
|
|
377
|
+
"sentinel_shape_invalid",
|
|
378
|
+
"capability_unresolved",
|
|
379
|
+
"service_unresolved",
|
|
380
|
+
"control_service_unresolved",
|
|
381
|
+
"process_name_duplicate",
|
|
382
|
+
"inputs_digest_mismatch",
|
|
383
|
+
"bundle_schema_unsupported",
|
|
384
|
+
"bundle_manifest_invalid",
|
|
385
|
+
"bundle_manifest_drifted",
|
|
386
|
+
"bundle_digest_mismatch",
|
|
387
|
+
"bundle_digest_invalid",
|
|
388
|
+
"inputs_digest_invalid",
|
|
389
|
+
"file_sha256_invalid",
|
|
390
|
+
"source_digest_invalid",
|
|
391
|
+
"bundle_file_missing",
|
|
392
|
+
"bundle_file_changed",
|
|
393
|
+
"bundle_file_unlisted",
|
|
394
|
+
"bundle_symlink_forbidden",
|
|
395
|
+
"bundle_path_unsafe",
|
|
396
|
+
"depends_on_unresolved",
|
|
397
|
+
"depends_on_cycle",
|
|
398
|
+
"seed_file_missing",
|
|
399
|
+
"seed_file_unlisted",
|
|
400
|
+
"process_count_exceeded",
|
|
401
|
+
"parallelism_out_of_range",
|
|
402
|
+
"evidence_seam_required",
|
|
403
|
+
"processes_required",
|
|
404
|
+
"processes_and_seed_forbidden",
|
|
405
|
+
"document_only_for_compose",
|
|
406
|
+
"compose_runtime_requires_document",
|
|
407
|
+
"build_command_step_empty",
|
|
408
|
+
"started_check_requires_exactly_one_of_port_or_log_marker",
|
|
409
|
+
"resolved_secret_forbidden",
|
|
410
|
+
"capability_slug_invalid",
|
|
411
|
+
"process_name_invalid",
|
|
412
|
+
"fixed_port_reserved",
|
|
413
|
+
}
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _bundle_unavailable_code(raw_message: str) -> str:
|
|
418
|
+
code = raw_message.split(":", 1)[0].strip()
|
|
419
|
+
return code if code in _SECTION_2E_CODES else "bundle_manifest_invalid"
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
class DefaultBundleSource:
|
|
423
|
+
"""Looks for an already-authored bundle at `work_directory / bundle_dir_name`. This is a
|
|
424
|
+
placeholder location this module invented (see the module-level STUCK DECISION note) — a real
|
|
425
|
+
bundle-authoring stage should either write there or be wired in via its own `BundleSource`."""
|
|
426
|
+
|
|
427
|
+
def __init__(self, bundle_dir_name: str = DEFAULT_BUNDLE_DIR_NAME) -> None:
|
|
428
|
+
self._bundle_dir_name = bundle_dir_name
|
|
429
|
+
|
|
430
|
+
def load(
|
|
431
|
+
self, job: HarnessJob, *, source: Path, work_directory: Path
|
|
432
|
+
) -> tuple[EnvironmentBundleV2, Path]:
|
|
433
|
+
del job, source # unused by the default (a real stage would author from these)
|
|
434
|
+
bundle_dir = work_directory / self._bundle_dir_name
|
|
435
|
+
try:
|
|
436
|
+
manifest = load_bundle_v2(bundle_dir)
|
|
437
|
+
except BundleV2Error as exc:
|
|
438
|
+
raise BundleUnavailableError(
|
|
439
|
+
_bundle_unavailable_code(exc.args[0]), str(exc)
|
|
440
|
+
) from exc
|
|
441
|
+
return manifest, bundle_dir
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
# =================================================================================================
|
|
445
|
+
# Scenario source -- generation is a separate contract (in review, not available here); the
|
|
446
|
+
# pre-allocation CALL is this module's (ScenariosClient below). Injectable for the same reason as
|
|
447
|
+
# BundleSource: the glue between "generated scenarios" and "pre-allocated against the platform" can
|
|
448
|
+
# only be finished once that contract's payload shape lands.
|
|
449
|
+
# =================================================================================================
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
class ScenarioSourceNotWired(RuntimeError):
|
|
453
|
+
"""The default `ScenarioSource` — no Scenario Generation Contract implementation exists in this
|
|
454
|
+
repo yet. Raised rather than fabricating scenarios, and mapped to FAILED / `platform_sync` /
|
|
455
|
+
`validating_scenarios`, matching spine §5 step 3.5's own failure mapping for a pre-allocation
|
|
456
|
+
that never completes."""
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
class ScenarioSource(Protocol):
|
|
460
|
+
async def build(
|
|
461
|
+
self,
|
|
462
|
+
job: HarnessJob,
|
|
463
|
+
bundle: EnvironmentBundleV2,
|
|
464
|
+
scenarios_client: "ScenariosClient",
|
|
465
|
+
*,
|
|
466
|
+
pool: WorldPool,
|
|
467
|
+
world_factory: WorldFactory,
|
|
468
|
+
bundle_dir: Path,
|
|
469
|
+
) -> Sequence[Scenario]: ...
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
class NotWiredScenarioSource:
|
|
473
|
+
async def build(
|
|
474
|
+
self,
|
|
475
|
+
job: HarnessJob,
|
|
476
|
+
bundle: EnvironmentBundleV2,
|
|
477
|
+
scenarios_client: "ScenariosClient",
|
|
478
|
+
*,
|
|
479
|
+
pool: WorldPool,
|
|
480
|
+
world_factory: WorldFactory,
|
|
481
|
+
bundle_dir: Path,
|
|
482
|
+
) -> Sequence[Scenario]:
|
|
483
|
+
del job, bundle, scenarios_client, pool, world_factory, bundle_dir
|
|
484
|
+
raise ScenarioSourceNotWired(
|
|
485
|
+
"no ScenarioSource wired -- scenario generation is not implemented in this repo yet "
|
|
486
|
+
"(Scenario Generation Contract, in review)"
|
|
487
|
+
)
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
# =================================================================================================
|
|
491
|
+
# WorldFactory -- real HostedWorld instances, fed by build.json's row counts (never
|
|
492
|
+
# a partial map).
|
|
493
|
+
# =================================================================================================
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
class WorldFactoryError(RuntimeError):
|
|
497
|
+
"""The provisioner handed back a runtime this factory cannot build a `World` for — a bug
|
|
498
|
+
upstream (no postgres endpoint despite §2e's `no_sql_store` guarantee, or `build.json` missing
|
|
499
|
+
the row counts for that store), never a scenario-code fault."""
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _process_runtime_error_domain(exc: ProcessRuntimeError) -> FailureDomain:
|
|
503
|
+
"""v1.15 §2f: the producer (`process_runtime.py`) resolves and carries `domain` at the raise
|
|
504
|
+
site -- read it directly rather than re-deriving `spawn_failed`'s managed/source split from
|
|
505
|
+
the manifest (the old approach could not tell which process kind failed without one). The
|
|
506
|
+
imported `SECTION_2F_DOMAIN` map is a fallback ONLY, for an error that reaches here with no
|
|
507
|
+
carried domain -- logged when it fires, matching the scheduler's own rule.
|
|
508
|
+
"""
|
|
509
|
+
if exc.domain is not None:
|
|
510
|
+
return exc.domain
|
|
511
|
+
if exc.code in SECTION_2F_DOMAIN:
|
|
512
|
+
logger.warning(
|
|
513
|
+
"process_runtime error %r crossed the §4 seam with no carried domain; using the §2f "
|
|
514
|
+
"fallback map (%s)",
|
|
515
|
+
exc.code,
|
|
516
|
+
SECTION_2F_DOMAIN[exc.code].value,
|
|
517
|
+
)
|
|
518
|
+
return SECTION_2F_DOMAIN[exc.code]
|
|
519
|
+
return FailureDomain.INFRASTRUCTURE # internal_* etc. -- the honest default
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
_SECTION_2F_CODES: frozenset[str] = frozenset(SECTION_2F_DOMAIN)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def _section_2f_code(code: str) -> str:
|
|
526
|
+
# §2f is closed (contract §4.6) -- `process_runtime.py`'s own `internal_*` codes, and this
|
|
527
|
+
# module's untyped-exception fallback, must never cross the outbound seam unlabeled, matching
|
|
528
|
+
# the discipline `_bundle_unavailable_code` already applies to §2e. The real code is
|
|
529
|
+
# still visible on the wire -- it stays in `message` (`ProcessRuntimeError.__str__` embeds it,
|
|
530
|
+
# and the untyped-exception call site prefixes it explicitly).
|
|
531
|
+
return code if code in _SECTION_2F_CODES else "spawn_failed"
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def _find_postgres_endpoint(runtime: EnvironmentRuntime) -> RuntimeEndpoint:
|
|
535
|
+
for endpoint in runtime.endpoints.values():
|
|
536
|
+
if endpoint.protocol == "postgres":
|
|
537
|
+
return endpoint
|
|
538
|
+
raise WorldFactoryError(
|
|
539
|
+
f"world {runtime.world_index}: no postgres-protocol endpoint in {sorted(runtime.endpoints)} "
|
|
540
|
+
"-- §2e's no_sql_store rule should make this unreachable"
|
|
541
|
+
)
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def load_build_output(work_directory: Path) -> dict[str, Any]:
|
|
545
|
+
"""`write_build_output` (process_runtime.py) writes `<work_directory>/artifacts/build.json`.
|
|
546
|
+
Read fresh each call — cheap, and the row counts are immutable after baseline freeze, so
|
|
547
|
+
re-reading is simpler than a cache invalidation story for the same modest cost."""
|
|
548
|
+
path = work_directory / "artifacts" / "build.json"
|
|
549
|
+
try:
|
|
550
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
551
|
+
except (OSError, ValueError) as exc:
|
|
552
|
+
raise WorldFactoryError(f"build.json unreadable at {path}: {exc}") from exc
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def row_counts_for_capability(
|
|
556
|
+
build_output: dict[str, Any], capability: str
|
|
557
|
+
) -> dict[str, int]:
|
|
558
|
+
for store in build_output.get("stores", []):
|
|
559
|
+
if store.get("capability") == capability:
|
|
560
|
+
counts = store.get("row_counts") or {}
|
|
561
|
+
return {str(name): int(count) for name, count in counts.items()}
|
|
562
|
+
raise WorldFactoryError(
|
|
563
|
+
f"build.json has no store entry for capability {capability!r} — the provisioner "
|
|
564
|
+
"guarantees a complete row-count map per store, so this bundle's build output is malformed"
|
|
565
|
+
)
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
class ProcessWorldFactory:
|
|
569
|
+
"""Builds a real `HostedWorld` over the runtime's postgres endpoint. `AttachedPostgresStore`
|
|
570
|
+
(not the bare `PostgresStore`) is the correct base here — it takes a raw DSN and never manages
|
|
571
|
+
a container's own lifecycle, matching a hosted world where `ProcessRuntimeProvider` already
|
|
572
|
+
owns the postgres process."""
|
|
573
|
+
|
|
574
|
+
def __init__(self, work_directory: Path) -> None:
|
|
575
|
+
self._work_directory = work_directory
|
|
576
|
+
|
|
577
|
+
async def create(self, runtime: EnvironmentRuntime, *, rng: random.Random) -> World:
|
|
578
|
+
endpoint = _find_postgres_endpoint(runtime)
|
|
579
|
+
build_output = await asyncio.to_thread(load_build_output, self._work_directory)
|
|
580
|
+
row_counts = row_counts_for_capability(build_output, endpoint.capability)
|
|
581
|
+
store = AttachedPostgresStore(endpoint.address)
|
|
582
|
+
return await asyncio.to_thread(
|
|
583
|
+
HostedWorld, store, runtime.world_index, rng, row_counts
|
|
584
|
+
)
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
# =================================================================================================
|
|
588
|
+
# CallRunner -- the real voice track. Explicit LiveKit jobs and auto-discovered voice contracts
|
|
589
|
+
# use it. Explicit Vapi/Retell jobs remain outside the repository-hosted runner.
|
|
590
|
+
# =================================================================================================
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
class CallRunnerNotWired(RuntimeError):
|
|
594
|
+
"""Raised by `NotWiredCallRunner`. `hosted_scheduler._execute` treats any exception out of
|
|
595
|
+
`CallRunner.run` (other than `WorldUnavailable`/`CallAborted`) as `call_failed`
|
|
596
|
+
(`FailureDomain.INFRASTRUCTURE`, retried once) — so a job run with nothing wired here degrades
|
|
597
|
+
every scenario to one retry-then-errored receipt rather than crashing the process."""
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
class NotWiredCallRunner:
|
|
601
|
+
async def run(
|
|
602
|
+
self,
|
|
603
|
+
scenario: Scenario,
|
|
604
|
+
runtime: EnvironmentRuntime,
|
|
605
|
+
*,
|
|
606
|
+
world: World | None = None,
|
|
607
|
+
) -> CallOutcome:
|
|
608
|
+
del scenario, runtime, world
|
|
609
|
+
raise CallRunnerNotWired(
|
|
610
|
+
"no CallRunner wired -- the live voice-simulation call runner is a separate track"
|
|
611
|
+
)
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
_VOICE_CONNECTORS = {"livekit", "vapi", "retell"}
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
def _bundle_contract_value(bundle_dir: Path, key: str) -> str | None:
|
|
618
|
+
path = bundle_dir / "contract.json"
|
|
619
|
+
if not path.is_file():
|
|
620
|
+
return None
|
|
621
|
+
try:
|
|
622
|
+
body = json.loads(path.read_text(encoding="utf-8"))
|
|
623
|
+
except (OSError, ValueError):
|
|
624
|
+
return None
|
|
625
|
+
if not isinstance(body, dict):
|
|
626
|
+
return None
|
|
627
|
+
value = str(body.get(key) or "").strip().lower()
|
|
628
|
+
return value or None
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
def _bundle_contract_modality(bundle_dir: Path) -> str | None:
|
|
632
|
+
return _bundle_contract_value(bundle_dir, "modality")
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def _default_build_call_runner(
|
|
636
|
+
adapter: "OutboundAdapter", context: CallRunnerContext
|
|
637
|
+
) -> CallRunner:
|
|
638
|
+
"""The real factory: `NotWiredCallRunner` stays exactly as documented for every connector
|
|
639
|
+
outside the LiveKit-dispatched voice path; a `"livekit"` job gets a real `CallRunnerImpl`,
|
|
640
|
+
whose OWN pre-dial validation (`call_runner._check_config`) is what surfaces an
|
|
641
|
+
incomplete-but-present config as a typed `call_failed`/infrastructure retry --
|
|
642
|
+
`capability_unavailable` stays unreachable from this seam (would require a scheduler edit;
|
|
643
|
+
the contract itself calls it "a follow-up, not shipped with this text")."""
|
|
644
|
+
connector = context.job.agent.connector.lower()
|
|
645
|
+
modality = _bundle_contract_modality(context.bundle_dir)
|
|
646
|
+
if connector == "retell_chat":
|
|
647
|
+
from .retell_chat_call_runner import RetellChatCallRunner
|
|
648
|
+
|
|
649
|
+
return RetellChatCallRunner(adapter, context)
|
|
650
|
+
if connector in _VOICE_CONNECTORS or (connector == "auto" and modality == "voice"):
|
|
651
|
+
# The understand stage read this off the agent's own instructions, so the contract is the
|
|
652
|
+
# only source. Carried through the process environment because `CallRunnerImpl` is handed a
|
|
653
|
+
# context and a scenario document, neither of which reaches the contract; this is an
|
|
654
|
+
# internal hop, not a knob, and nothing outside sets it.
|
|
655
|
+
declared = _bundle_contract_value(context.bundle_dir, "call_direction")
|
|
656
|
+
if declared:
|
|
657
|
+
os.environ["ALK_CALL_DIRECTION"] = declared
|
|
658
|
+
return CallRunnerImpl(adapter, context)
|
|
659
|
+
# Repository-hosted text targets advertise their concrete HTTP interface in the frozen
|
|
660
|
+
# contract adopted into Bundle V2. Connector-only Vapi/Retell remains on the existing
|
|
661
|
+
# NotWired path and is deliberately not inferred as repository chat.
|
|
662
|
+
if (context.bundle_dir / "contract.json").is_file():
|
|
663
|
+
from .chat_call_runner import HostedChatCallRunner
|
|
664
|
+
|
|
665
|
+
return HostedChatCallRunner(adapter, context)
|
|
666
|
+
return NotWiredCallRunner()
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
# =================================================================================================
|
|
670
|
+
# Scenario pre-allocation -- a thin client against endpoints.scenarios (outbound-channels.md v1.3
|
|
671
|
+
# Authentication: bearer + X-Harness-Fence, `{"result": {...}}` envelope, job-scoped idempotent).
|
|
672
|
+
# Previously unowned; owned by this module now.
|
|
673
|
+
# =================================================================================================
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
class ScenarioPreallocationError(RuntimeError):
|
|
677
|
+
def __init__(self, error: ob.ChannelError | None) -> None:
|
|
678
|
+
self.error = error
|
|
679
|
+
super().__init__(
|
|
680
|
+
"scenario pre-allocation failed" if error is None else error.message
|
|
681
|
+
)
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
class ScenariosClient:
|
|
685
|
+
"""RESOLVED (p13-worker-r2, reports/p13-worker-r2.md CONTRACT NOTES): the Scenario
|
|
686
|
+
Generation Contract (PR #63) documented two paths (`run-tests/provision/` +
|
|
687
|
+
`run-tests/{id}/test-executions/`) and a position-ordered `scenario_ids` response, but the
|
|
688
|
+
platform's actual, live route (futureagi/simulate/views/hosted_harness.py:78-90,
|
|
689
|
+
urls.py:128-132) mints exactly ONE url per attempt -- a DRF detail `@action` with no
|
|
690
|
+
`url_path`, so the router only ever produces `.../scenarios/`, never a `provision/`/`begin/`
|
|
691
|
+
sub-resource. The real dispatch key is a body-level `operation: "provision"|"begin"` field
|
|
692
|
+
(serializers/hosted_harness.py:201-226's `HarnessScenarioOperationSerializer`). This class's
|
|
693
|
+
transport (`_post`) is unchanged -- `provision_path`/`begin_path` are the SAME
|
|
694
|
+
constructor-injectable placeholders as before, now correctly defaulted to an EMPTY suffix (the
|
|
695
|
+
real route needs none) rather than a guessed path segment; `register_with_platform`
|
|
696
|
+
(scenario_source.py) is what adds the `operation` field into each payload before calling
|
|
697
|
+
`.provision()`/`.begin()`, matching this class's existing "operation field in payload" seam
|
|
698
|
+
rather than requiring a change to either method's body. Shares `channel_state` with the other
|
|
699
|
+
three channels (a fence on any one must stop all of them, per outbound.py's own `ChannelState`
|
|
700
|
+
docstring)."""
|
|
701
|
+
|
|
702
|
+
def __init__(
|
|
703
|
+
self,
|
|
704
|
+
capabilities: ob.HostedCapabilities,
|
|
705
|
+
transport: ob.Transport | None = None,
|
|
706
|
+
*,
|
|
707
|
+
retry_policy: ob.RetryPolicy | None = None,
|
|
708
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
709
|
+
rng: Callable[[], float] = random.random,
|
|
710
|
+
channel_state: ob.ChannelState | None = None,
|
|
711
|
+
provision_path: str = "",
|
|
712
|
+
begin_path: str = "",
|
|
713
|
+
) -> None:
|
|
714
|
+
self._capabilities = capabilities
|
|
715
|
+
self._transport = transport or ob.RequestsTransport()
|
|
716
|
+
self._retry_policy = retry_policy or ob.RetryPolicy()
|
|
717
|
+
self._sleep = sleep
|
|
718
|
+
self._rng = rng
|
|
719
|
+
self._channel_state = channel_state or ob.ChannelState()
|
|
720
|
+
self._provision_path = provision_path
|
|
721
|
+
self._begin_path = begin_path
|
|
722
|
+
|
|
723
|
+
def provision(
|
|
724
|
+
self, payload: dict[str, Any], *, deadline: float | None = None
|
|
725
|
+
) -> dict[str, Any]:
|
|
726
|
+
return self._post(self._provision_path, payload, deadline=deadline)
|
|
727
|
+
|
|
728
|
+
def begin(
|
|
729
|
+
self, payload: dict[str, Any], *, deadline: float | None = None
|
|
730
|
+
) -> dict[str, Any]:
|
|
731
|
+
return self._post(self._begin_path, payload, deadline=deadline)
|
|
732
|
+
|
|
733
|
+
def _post(
|
|
734
|
+
self, path_suffix: str, payload: dict[str, Any], *, deadline: float | None
|
|
735
|
+
) -> dict[str, Any]:
|
|
736
|
+
self._channel_state.check()
|
|
737
|
+
url = f"{self._capabilities.endpoints.scenarios}{path_suffix}"
|
|
738
|
+
|
|
739
|
+
def perform(_attempt: int) -> ob.TransportResponse:
|
|
740
|
+
return self._transport.request(
|
|
741
|
+
"POST",
|
|
742
|
+
url,
|
|
743
|
+
headers=self._capabilities.auth_headers(),
|
|
744
|
+
json_body=payload,
|
|
745
|
+
)
|
|
746
|
+
|
|
747
|
+
try:
|
|
748
|
+
response, error = ob._perform_with_retry(
|
|
749
|
+
perform,
|
|
750
|
+
retry_policy=self._retry_policy,
|
|
751
|
+
sleep=self._sleep,
|
|
752
|
+
rng=self._rng,
|
|
753
|
+
deadline=deadline,
|
|
754
|
+
)
|
|
755
|
+
except (ob.HostedFencedError, ob.HostedChannelFailedError) as exc:
|
|
756
|
+
self._channel_state.latch(exc)
|
|
757
|
+
raise
|
|
758
|
+
if error is not None or response is None:
|
|
759
|
+
raise ScenarioPreallocationError(error)
|
|
760
|
+
body = response.body if isinstance(response.body, dict) else {}
|
|
761
|
+
result = body.get("result")
|
|
762
|
+
if not isinstance(result, dict):
|
|
763
|
+
raise ScenarioPreallocationError(
|
|
764
|
+
ob.ChannelError(
|
|
765
|
+
ob.ChannelOutcome.PERMANENT_ITEM,
|
|
766
|
+
FailureDomain.PLATFORM_SYNC,
|
|
767
|
+
"scenarios_envelope_invalid",
|
|
768
|
+
"response body has no {'result': {...}} envelope",
|
|
769
|
+
)
|
|
770
|
+
)
|
|
771
|
+
return result
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
# =================================================================================================
|
|
775
|
+
# OutboundPort adapter -- the real emit pipeline: redact -> capabilities.event_builder() ->
|
|
776
|
+
# spool.append -> EventsClient.flush(). Also: baseline_frozen/parallelism_degraded from build.json,
|
|
777
|
+
# terminal events (exactly one, last), artifact-before-receipt ordering,
|
|
778
|
+
# and RunResult.aborted -> TerminalFailure(infrastructure, running, "world_pool_exhausted").
|
|
779
|
+
# =================================================================================================
|
|
780
|
+
|
|
781
|
+
|
|
782
|
+
_TERMINAL_FAILURE_MESSAGE_MAX_CHARS = 4096 # an unbounded `failure.message` can blow
|
|
783
|
+
# EVENT_PAYLOAD_MAX_BYTES and hard-reject the WHOLE terminal event; log is the only event type
|
|
784
|
+
# that self-truncates. 4KB is ample for a diagnostic message.
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
def _cap_failure_message(message: str) -> str:
|
|
788
|
+
if len(message) <= _TERMINAL_FAILURE_MESSAGE_MAX_CHARS:
|
|
789
|
+
return message
|
|
790
|
+
marker = "…[truncated]"
|
|
791
|
+
return message[: _TERMINAL_FAILURE_MESSAGE_MAX_CHARS - len(marker)] + marker
|
|
792
|
+
|
|
793
|
+
|
|
794
|
+
# guest-side mirror of outbound-channels.md's artifact level table (Channel 3) -- no module
|
|
795
|
+
# owns this table yet (the sealer's own version lives at `artifacts.py::seal_artifacts`, scoped to
|
|
796
|
+
# the local-SDK path); this hosted upload path needs its own "guest enforces it first" half.
|
|
797
|
+
_ARTIFACT_LEVEL_FORBIDDEN_KINDS: dict[ArtifactLevel, frozenset[ob.ArtifactKind]] = {
|
|
798
|
+
ArtifactLevel.METADATA_ONLY: frozenset(
|
|
799
|
+
{
|
|
800
|
+
ob.ArtifactKind.RECORDING_COMBINED,
|
|
801
|
+
ob.ArtifactKind.RECORDING_STEREO,
|
|
802
|
+
ob.ArtifactKind.RECORDING_CUSTOMER,
|
|
803
|
+
ob.ArtifactKind.RECORDING_ASSISTANT,
|
|
804
|
+
ob.ArtifactKind.TRACE,
|
|
805
|
+
ob.ArtifactKind.TOOL_TRACE,
|
|
806
|
+
ob.ArtifactKind.TRANSCRIPT,
|
|
807
|
+
ob.ArtifactKind.OTHER,
|
|
808
|
+
}
|
|
809
|
+
),
|
|
810
|
+
ArtifactLevel.TRACES: frozenset(
|
|
811
|
+
{
|
|
812
|
+
ob.ArtifactKind.RECORDING_COMBINED,
|
|
813
|
+
ob.ArtifactKind.RECORDING_STEREO,
|
|
814
|
+
ob.ArtifactKind.RECORDING_CUSTOMER,
|
|
815
|
+
ob.ArtifactKind.RECORDING_ASSISTANT,
|
|
816
|
+
ob.ArtifactKind.OTHER,
|
|
817
|
+
}
|
|
818
|
+
),
|
|
819
|
+
ArtifactLevel.TRACES_AND_RECORDINGS: frozenset({ob.ArtifactKind.OTHER}),
|
|
820
|
+
ArtifactLevel.FULL: frozenset(),
|
|
821
|
+
# `local-only` is rejected at hosted admission (`local_only_not_hosted`) per the contract --
|
|
822
|
+
# this adapter should never see it for a hosted job; forbid everything as a defensive default.
|
|
823
|
+
ArtifactLevel.LOCAL_ONLY: frozenset(ob.ArtifactKind),
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
class OutboundAdapter:
|
|
828
|
+
"""Implements `hosted_scheduler.OutboundPort` plus the extra surface the entrypoint itself
|
|
829
|
+
needs (`stage_changed`, `baseline_frozen`, `parallelism_degraded`, `upload_artifact`,
|
|
830
|
+
`push_manifest`, `emit_terminal`) — hosted_scheduler.py only names the five methods scenario
|
|
831
|
+
code needs; everything else here is this module's own.
|
|
832
|
+
|
|
833
|
+
Fencing (HostedFencedError) is caught INTERNALLY by every method, never re-raised: letting it
|
|
834
|
+
escape into `HostedScheduler._emit()` (which catches bare `Exception` and tries to log through
|
|
835
|
+
the very port that just raised) would silently swallow the fence and let the scheduler keep
|
|
836
|
+
working an attempt that can no longer report anything. `is_fenced` is the flag the entrypoint's
|
|
837
|
+
orchestration (and `cancel_requested`) polls instead.
|
|
838
|
+
"""
|
|
839
|
+
|
|
840
|
+
def __init__(
|
|
841
|
+
self,
|
|
842
|
+
capabilities: ob.HostedCapabilities,
|
|
843
|
+
*,
|
|
844
|
+
events_spool: ob.OutboundSpool,
|
|
845
|
+
events_client: ob.EventsClient,
|
|
846
|
+
results_client: ob.ResultsClient,
|
|
847
|
+
artifacts_client: ob.ArtifactsClient,
|
|
848
|
+
channel_state: ob.ChannelState,
|
|
849
|
+
extra_secret_values: tuple[str, ...] = (),
|
|
850
|
+
clock: Callable[[], datetime] = lambda: datetime.now(timezone.utc),
|
|
851
|
+
flush_window_seconds: float = ob.FLUSH_WINDOW_SECONDS,
|
|
852
|
+
) -> None:
|
|
853
|
+
self._capabilities = capabilities
|
|
854
|
+
self._spool = events_spool
|
|
855
|
+
self._events = events_client
|
|
856
|
+
self._results = results_client
|
|
857
|
+
self._artifacts = artifacts_client
|
|
858
|
+
self._channel_state = channel_state
|
|
859
|
+
self._extra_secret_values = extra_secret_values
|
|
860
|
+
self._clock = clock
|
|
861
|
+
# `event_builder`'s own `extra_secret_values` binding is what lets
|
|
862
|
+
# `build_event_record` redact `log.message`/`world_unhealthy.cause`/
|
|
863
|
+
# `baseline_frozen.baseline_ref`/`terminal.failure.{code,message}` for every event this
|
|
864
|
+
# adapter emits -- binding it here, alongside identity, gives Channel 1 full redaction coverage.
|
|
865
|
+
self._event_builder = capabilities.event_builder(
|
|
866
|
+
extra_secret_values=extra_secret_values
|
|
867
|
+
)
|
|
868
|
+
self._stage_started = False
|
|
869
|
+
self._current_stage = HarnessStage.QUEUED
|
|
870
|
+
self._uploaded_digests: set[str] = set()
|
|
871
|
+
self._manifest_entries: list[dict[str, Any]] = []
|
|
872
|
+
self._terminal_emitted = False
|
|
873
|
+
# §0.6 v1.14 (exit code 4): the terminal record's own spool sequence, and whether the
|
|
874
|
+
# platform ever permanently rejected it by name -- `terminal_undelivered` (below) needs to
|
|
875
|
+
# tell "this specific record landed" apart from "some flush somewhere failed."
|
|
876
|
+
self._terminal_sequence: int | None = None
|
|
877
|
+
self._terminal_rejected = False
|
|
878
|
+
self._scenario_counts: dict[str, int] = {
|
|
879
|
+
"passed": 0,
|
|
880
|
+
"failed": 0,
|
|
881
|
+
"errored": 0,
|
|
882
|
+
"skipped": 0,
|
|
883
|
+
}
|
|
884
|
+
self._fenced_error: Exception | None = None
|
|
885
|
+
self._channel_failed_error: Exception | None = None
|
|
886
|
+
# the 120s flush window (§5.5) -- armed once, at whichever comes first: a cancel
|
|
887
|
+
# signal (`arm_flush_window` called explicitly by `run_job`'s `cancel_requested`) or the
|
|
888
|
+
# terminal event (`emit_terminal` below arms it itself, so no caller can forget).
|
|
889
|
+
self._flush_window_seconds = flush_window_seconds
|
|
890
|
+
self._flush_window_start: float | None = None
|
|
891
|
+
# job.artifacts is only known once job.json is parsed, which happens after this
|
|
892
|
+
# adapter is built (capabilities load, and the "no channel on a capabilities failure"
|
|
893
|
+
# contract, must come first) -- `configure_artifacts` below is called once it's available;
|
|
894
|
+
# this default is never actually exercised in practice, just a safe placeholder shape.
|
|
895
|
+
self._artifacts_policy = HarnessArtifactPolicy()
|
|
896
|
+
# `recording_headroom_bytes` stays 0 -- this adapter has no visibility into how many
|
|
897
|
+
# scenarios are still to run (or how large their recordings will be) at construction time,
|
|
898
|
+
# unlike the scheduler; sizing it here would be a guess dressed up as enforcement.
|
|
899
|
+
self._budget_tracker = ob.ArtifactBudgetTracker(
|
|
900
|
+
self._artifacts_policy.max_artifact_bytes
|
|
901
|
+
)
|
|
902
|
+
# `would_admit` (check) and `record` (reserve) must run as one atomic step -- two
|
|
903
|
+
# concurrent scenarios at W>1 racing the same remaining budget could otherwise both pass
|
|
904
|
+
# the check against a snapshot neither has updated yet.
|
|
905
|
+
self._artifact_budget_lock = asyncio.Lock()
|
|
906
|
+
|
|
907
|
+
@property
|
|
908
|
+
def is_fenced(self) -> bool:
|
|
909
|
+
return self._fenced_error is not None
|
|
910
|
+
|
|
911
|
+
@property
|
|
912
|
+
def terminal_undelivered(self) -> bool:
|
|
913
|
+
"""The terminal was spooled (`emit_terminal` succeeded) but never confirmed delivered: the
|
|
914
|
+
platform permanently rejected the terminal item by name, or the spool's watermark never
|
|
915
|
+
reached the terminal's own sequence at all (channel exhaustion, a dead channel, or the
|
|
916
|
+
flush window running out before delivery). Exit 0 would claim a flush that provably never
|
|
917
|
+
happened. Fencing is checked by the caller first and always wins -- once fenced, whether
|
|
918
|
+
the terminal was ALSO undelivered is moot."""
|
|
919
|
+
if self._terminal_sequence is None or self.is_fenced:
|
|
920
|
+
return False
|
|
921
|
+
return (
|
|
922
|
+
self._terminal_rejected or self._spool.watermark() < self._terminal_sequence
|
|
923
|
+
)
|
|
924
|
+
|
|
925
|
+
@property
|
|
926
|
+
def scenario_counts(self) -> dict[str, int]:
|
|
927
|
+
return dict(self._scenario_counts)
|
|
928
|
+
|
|
929
|
+
def configure_artifacts(self, policy: HarnessArtifactPolicy) -> None:
|
|
930
|
+
self._artifacts_policy = policy
|
|
931
|
+
self._budget_tracker = ob.ArtifactBudgetTracker(policy.max_artifact_bytes)
|
|
932
|
+
|
|
933
|
+
def arm_flush_window(self) -> None:
|
|
934
|
+
if self._flush_window_start is None:
|
|
935
|
+
self._flush_window_start = time.monotonic()
|
|
936
|
+
|
|
937
|
+
def deadline(self) -> float | None:
|
|
938
|
+
if self._flush_window_start is None:
|
|
939
|
+
return None
|
|
940
|
+
return self._flush_window_start + self._flush_window_seconds
|
|
941
|
+
|
|
942
|
+
def _record_channel_error(self, exc: Exception) -> None:
|
|
943
|
+
if isinstance(exc, ob.HostedFencedError):
|
|
944
|
+
self._fenced_error = self._fenced_error or exc
|
|
945
|
+
else:
|
|
946
|
+
self._channel_failed_error = self._channel_failed_error or exc
|
|
947
|
+
logger.error("outbound channel latched: %s", exc)
|
|
948
|
+
|
|
949
|
+
def _guarded(self, fn: Callable[[], Any]) -> Any:
|
|
950
|
+
"""Runs one outbound client call. Catches `HostedFencedError`/`HostedChannelFailedError`
|
|
951
|
+
so neither escapes as an ordinary exception (see class docstring)."""
|
|
952
|
+
try:
|
|
953
|
+
self._channel_state.check()
|
|
954
|
+
except (
|
|
955
|
+
ob.HostedFencedError,
|
|
956
|
+
ob.HostedChannelFailedError,
|
|
957
|
+
ob.HostedAttemptSupersededError,
|
|
958
|
+
) as exc:
|
|
959
|
+
self._record_channel_error(exc)
|
|
960
|
+
return None
|
|
961
|
+
try:
|
|
962
|
+
return fn()
|
|
963
|
+
except (ob.HostedFencedError, ob.HostedChannelFailedError) as exc:
|
|
964
|
+
self._channel_state.latch(exc)
|
|
965
|
+
self._record_channel_error(exc)
|
|
966
|
+
return None
|
|
967
|
+
|
|
968
|
+
# -- events -------------------------------------------------------------------------------
|
|
969
|
+
|
|
970
|
+
def _emit_event(
|
|
971
|
+
self,
|
|
972
|
+
*,
|
|
973
|
+
stage: HarnessStage,
|
|
974
|
+
type_: ob.OutboundEventType,
|
|
975
|
+
payload: dict[str, Any],
|
|
976
|
+
) -> None:
|
|
977
|
+
if self.is_fenced:
|
|
978
|
+
return # "stop emitting" -- no event of any type once fenced.
|
|
979
|
+
if self._terminal_emitted:
|
|
980
|
+
# `_bounded_close()` runs after the terminal is spooled -- an in-flight reconcile
|
|
981
|
+
# inside it can still call back into world_unhealthy/log. v1.3's "terminal ... exactly
|
|
982
|
+
# one, last emitted" is a hard invariant: anything after it is dropped locally, not
|
|
983
|
+
# spooled, rather than silently landing after the event the platform already finalized on.
|
|
984
|
+
# This also drops the rejected-event error log for anything the platform rejects on the
|
|
985
|
+
# SAME flush that carries the terminal -- diagnostic-quality only, since the rejected
|
|
986
|
+
# bytes are still recoverable as a `log`-kind artifact.
|
|
987
|
+
logger.warning(
|
|
988
|
+
"outbound event dropped after terminal: type=%s stage=%s",
|
|
989
|
+
type_.value,
|
|
990
|
+
stage.value,
|
|
991
|
+
)
|
|
992
|
+
return
|
|
993
|
+
event_id = f"event_{uuid.uuid4().hex}"
|
|
994
|
+
record = self._event_builder(
|
|
995
|
+
event_id=event_id,
|
|
996
|
+
emitted_at=self._clock(),
|
|
997
|
+
stage=stage,
|
|
998
|
+
type=type_,
|
|
999
|
+
payload=payload,
|
|
1000
|
+
)
|
|
1001
|
+
spooled = self._spool.append(record)
|
|
1002
|
+
if type_ is ob.OutboundEventType.TERMINAL:
|
|
1003
|
+
self._terminal_sequence = spooled.sequence
|
|
1004
|
+
self._current_stage = stage
|
|
1005
|
+
|
|
1006
|
+
async def _aemit_event(
|
|
1007
|
+
self,
|
|
1008
|
+
*,
|
|
1009
|
+
stage: HarnessStage,
|
|
1010
|
+
type_: ob.OutboundEventType,
|
|
1011
|
+
payload: dict[str, Any],
|
|
1012
|
+
) -> None:
|
|
1013
|
+
# the spool append fsyncs the file AND its directory -- routed off the event loop so
|
|
1014
|
+
# it never stalls every other concurrently-running scenario at W>1.
|
|
1015
|
+
await asyncio.to_thread(
|
|
1016
|
+
self._emit_event, stage=stage, type_=type_, payload=payload
|
|
1017
|
+
)
|
|
1018
|
+
|
|
1019
|
+
def stage_changed(self, to: HarnessStage) -> None:
|
|
1020
|
+
frm = self._current_stage.value if self._stage_started else None
|
|
1021
|
+
self._stage_started = True
|
|
1022
|
+
observability.stage(to.value)
|
|
1023
|
+
self._emit_event(
|
|
1024
|
+
stage=to,
|
|
1025
|
+
type_=ob.OutboundEventType.STAGE_CHANGED,
|
|
1026
|
+
payload={"from": frm, "to": to.value},
|
|
1027
|
+
)
|
|
1028
|
+
|
|
1029
|
+
def baseline_frozen(self, *, inputs_digest: str, baseline_ref: str) -> None:
|
|
1030
|
+
self._emit_event(
|
|
1031
|
+
stage=HarnessStage.VALIDATING_ENVIRONMENT,
|
|
1032
|
+
type_=ob.OutboundEventType.BASELINE_FROZEN,
|
|
1033
|
+
payload={"inputs_digest": inputs_digest, "baseline_ref": baseline_ref},
|
|
1034
|
+
)
|
|
1035
|
+
|
|
1036
|
+
def parallelism_degraded(
|
|
1037
|
+
self, *, requested: int, effective: int, reason: str
|
|
1038
|
+
) -> None:
|
|
1039
|
+
self._emit_event(
|
|
1040
|
+
stage=HarnessStage.VALIDATING_ENVIRONMENT,
|
|
1041
|
+
type_=ob.OutboundEventType.PARALLELISM_DEGRADED,
|
|
1042
|
+
payload={"requested": requested, "effective": effective, "reason": reason},
|
|
1043
|
+
)
|
|
1044
|
+
|
|
1045
|
+
def flush_events(
|
|
1046
|
+
self, *, deadline: float | None = None
|
|
1047
|
+
) -> ob.EventsFlushResult | None:
|
|
1048
|
+
return self._guarded(lambda: self._events.flush(deadline=deadline))
|
|
1049
|
+
|
|
1050
|
+
async def aflush_events(self, *, deadline: float | None = None) -> None:
|
|
1051
|
+
result = await asyncio.to_thread(self.flush_events, deadline=deadline)
|
|
1052
|
+
# a rejected event's own payload never reaches the platform any other way -- surface
|
|
1053
|
+
# it via a `log` event (error) and keep the bytes recoverable as a `log`-kind artifact,
|
|
1054
|
+
# keyed off `dropped_records` (captured before the spool physically drops them).
|
|
1055
|
+
if result is None or not result.rejected:
|
|
1056
|
+
return
|
|
1057
|
+
if self._terminal_sequence is not None and any(
|
|
1058
|
+
entry.get("sequence") == self._terminal_sequence
|
|
1059
|
+
for entry in result.rejected
|
|
1060
|
+
):
|
|
1061
|
+
# A permanent-item rejection is never retried -- the spool physically drops the record,
|
|
1062
|
+
# so no later flush can ever redeliver it.
|
|
1063
|
+
self._terminal_rejected = True
|
|
1064
|
+
dropped_by_sequence = {
|
|
1065
|
+
record.sequence: record for record in result.dropped_records
|
|
1066
|
+
}
|
|
1067
|
+
for entry in result.rejected:
|
|
1068
|
+
sequence = entry.get("sequence")
|
|
1069
|
+
await self.log(
|
|
1070
|
+
level="error",
|
|
1071
|
+
message=(
|
|
1072
|
+
f"event sequence={sequence} rejected by the platform: "
|
|
1073
|
+
f"{entry.get('code', 'unknown')}: {entry.get('message', '')}"
|
|
1074
|
+
),
|
|
1075
|
+
)
|
|
1076
|
+
record = dropped_by_sequence.get(sequence)
|
|
1077
|
+
if record is not None:
|
|
1078
|
+
await self.upload_artifact(record.body, kind=ob.ArtifactKind.LOG)
|
|
1079
|
+
|
|
1080
|
+
# -- OutboundPort (hosted_scheduler.py) ----------------------------------------------------
|
|
1081
|
+
|
|
1082
|
+
async def scenario_started(
|
|
1083
|
+
self, *, scenario_key: str, world_index: int, scenario_attempt: int
|
|
1084
|
+
) -> None:
|
|
1085
|
+
await self._aemit_event(
|
|
1086
|
+
stage=HarnessStage.RUNNING,
|
|
1087
|
+
type_=ob.OutboundEventType.SCENARIO_STARTED,
|
|
1088
|
+
payload={
|
|
1089
|
+
"scenario_key": scenario_key,
|
|
1090
|
+
"world_index": world_index,
|
|
1091
|
+
"scenario_attempt": scenario_attempt,
|
|
1092
|
+
},
|
|
1093
|
+
)
|
|
1094
|
+
|
|
1095
|
+
async def scenario_retried(
|
|
1096
|
+
self, *, scenario_key: str, from_world: int, to_world: int
|
|
1097
|
+
) -> None:
|
|
1098
|
+
await self._aemit_event(
|
|
1099
|
+
stage=HarnessStage.RUNNING,
|
|
1100
|
+
type_=ob.OutboundEventType.SCENARIO_RETRIED,
|
|
1101
|
+
payload={
|
|
1102
|
+
"scenario_key": scenario_key,
|
|
1103
|
+
"from_world": from_world,
|
|
1104
|
+
"to_world": to_world,
|
|
1105
|
+
},
|
|
1106
|
+
)
|
|
1107
|
+
|
|
1108
|
+
async def world_unhealthy(self, *, world_index: int, cause: str) -> None:
|
|
1109
|
+
# world_unhealthy.cause <=200 (WorldUnhealthyPayload hard-rejects over that, so this must
|
|
1110
|
+
# truncate BEFORE `_aemit_event`, not rely on the builder's own redaction, which runs
|
|
1111
|
+
# after this call and could not shrink an already-too-long string back into budget).
|
|
1112
|
+
redacted = ob.redact_outbound_text(cause, self._extra_secret_values)
|
|
1113
|
+
if len(redacted) > 200:
|
|
1114
|
+
redacted = redacted[:200]
|
|
1115
|
+
await self._aemit_event(
|
|
1116
|
+
stage=HarnessStage.RUNNING,
|
|
1117
|
+
type_=ob.OutboundEventType.WORLD_UNHEALTHY,
|
|
1118
|
+
payload={"world_index": world_index, "cause": redacted},
|
|
1119
|
+
)
|
|
1120
|
+
|
|
1121
|
+
async def log(self, *, level: str, message: str) -> None:
|
|
1122
|
+
await self._aemit_event(
|
|
1123
|
+
stage=self._current_stage,
|
|
1124
|
+
type_=ob.OutboundEventType.LOG,
|
|
1125
|
+
payload={"level": level, "message": message},
|
|
1126
|
+
)
|
|
1127
|
+
|
|
1128
|
+
async def receipt(self, receipt: ResultReceipt) -> None:
|
|
1129
|
+
if self.is_fenced:
|
|
1130
|
+
return
|
|
1131
|
+
# counted only once a push is actually attempted -- counting before this point would
|
|
1132
|
+
# include receipts that were never pushed (and the counts feed the terminal payload).
|
|
1133
|
+
self._scenario_counts[receipt.status] = (
|
|
1134
|
+
self._scenario_counts.get(receipt.status, 0) + 1
|
|
1135
|
+
)
|
|
1136
|
+
call: dict[str, Any] | None = None
|
|
1137
|
+
if receipt.call is not None and receipt.call.started_at is not None:
|
|
1138
|
+
transcript_artifact = receipt.call.transcript_artifact
|
|
1139
|
+
if transcript_artifact is not None:
|
|
1140
|
+
bare = transcript_artifact.split(":", 1)[-1]
|
|
1141
|
+
if bare not in self._uploaded_digests:
|
|
1142
|
+
# null it rather than shipping a receipt the platform will 422
|
|
1143
|
+
# (`artifact_unknown`) wholesale -- the contract explicitly blesses a null
|
|
1144
|
+
# `transcript_artifact` "named in a log event."
|
|
1145
|
+
await self.log(
|
|
1146
|
+
level="error",
|
|
1147
|
+
message=(
|
|
1148
|
+
f"receipt for {receipt.scenario_key} references un-acked transcript "
|
|
1149
|
+
f"artifact {transcript_artifact}; nulling it"
|
|
1150
|
+
),
|
|
1151
|
+
)
|
|
1152
|
+
transcript_artifact = None
|
|
1153
|
+
recording_artifacts: list[str] = []
|
|
1154
|
+
for artifact_id in receipt.call.recording_artifacts:
|
|
1155
|
+
bare = artifact_id.split(":", 1)[-1] if artifact_id else None
|
|
1156
|
+
if artifact_id and bare not in self._uploaded_digests:
|
|
1157
|
+
await self.log(
|
|
1158
|
+
level="error",
|
|
1159
|
+
message=(
|
|
1160
|
+
f"receipt for {receipt.scenario_key} references un-acked recording "
|
|
1161
|
+
f"artifact {artifact_id}; dropping it"
|
|
1162
|
+
),
|
|
1163
|
+
)
|
|
1164
|
+
continue
|
|
1165
|
+
recording_artifacts.append(artifact_id)
|
|
1166
|
+
ended_at = receipt.call.ended_at
|
|
1167
|
+
if ended_at is None:
|
|
1168
|
+
# outbound.CallSummary.ended_at is a required str -- a call that started but
|
|
1169
|
+
# never finished (CallAborted's partial) would otherwise fail build_result_receipt's
|
|
1170
|
+
# validation and silently drop the whole receipt (HostedScheduler._emit's blanket
|
|
1171
|
+
# except swallows it).
|
|
1172
|
+
ended_at = receipt.call.started_at
|
|
1173
|
+
await self.log(
|
|
1174
|
+
level="warning",
|
|
1175
|
+
message=(
|
|
1176
|
+
f"receipt for {receipt.scenario_key} has no call.ended_at; substituting "
|
|
1177
|
+
"started_at"
|
|
1178
|
+
),
|
|
1179
|
+
)
|
|
1180
|
+
call = {
|
|
1181
|
+
"started_at": receipt.call.started_at,
|
|
1182
|
+
"ended_at": ended_at,
|
|
1183
|
+
"duration_ms": receipt.call.duration_ms,
|
|
1184
|
+
"turns": receipt.call.turns,
|
|
1185
|
+
"transcript_artifact": transcript_artifact,
|
|
1186
|
+
"recording_artifacts": recording_artifacts,
|
|
1187
|
+
}
|
|
1188
|
+
stop_reason = getattr(receipt.call, "stop_reason", None)
|
|
1189
|
+
if stop_reason:
|
|
1190
|
+
call["stop_reason"] = stop_reason
|
|
1191
|
+
elif receipt.call is not None:
|
|
1192
|
+
# `hosted_scheduler.CallSummary.started_at` is `str | None`, but
|
|
1193
|
+
# `outbound.CallSummary.started_at` requires a real timestamp -- per the contract a
|
|
1194
|
+
# call summary is only present once the call has genuinely started, so a call that
|
|
1195
|
+
# never started is omitted here rather than shipped with a value that would fail
|
|
1196
|
+
# `build_result_receipt`'s own validation.
|
|
1197
|
+
await self.log(
|
|
1198
|
+
level="warning",
|
|
1199
|
+
message=f"receipt for {receipt.scenario_key} has no call.started_at; omitting call",
|
|
1200
|
+
)
|
|
1201
|
+
failure: dict[str, Any] | None = None
|
|
1202
|
+
if receipt.failure is not None:
|
|
1203
|
+
# Redact before capping -- truncating first can cut a secret in half at
|
|
1204
|
+
# the boundary and leave exact-substring redaction unable to find the surviving piece.
|
|
1205
|
+
redacted_failure_message = ob.redact_outbound_text(
|
|
1206
|
+
receipt.failure.message, self._extra_secret_values
|
|
1207
|
+
)
|
|
1208
|
+
failure = {
|
|
1209
|
+
"domain": receipt.failure.domain,
|
|
1210
|
+
"stage": receipt.failure.stage,
|
|
1211
|
+
"code": receipt.failure.code,
|
|
1212
|
+
"message": _cap_failure_message(redacted_failure_message),
|
|
1213
|
+
}
|
|
1214
|
+
wire = ob.build_result_receipt(
|
|
1215
|
+
job_id=self._capabilities.job_id,
|
|
1216
|
+
attempt_id=self._capabilities.attempt_id,
|
|
1217
|
+
attempt_number=self._capabilities.attempt_number,
|
|
1218
|
+
scenario_key=receipt.scenario_key,
|
|
1219
|
+
scenario_id=receipt.scenario_id,
|
|
1220
|
+
scenario_attempt=receipt.scenario_attempt,
|
|
1221
|
+
world_index=receipt.world_index,
|
|
1222
|
+
status=receipt.status,
|
|
1223
|
+
sub_goals=[
|
|
1224
|
+
{"name": g.name, "held": g.held, "reason": g.reason, "judged": g.judged}
|
|
1225
|
+
for g in receipt.sub_goals
|
|
1226
|
+
],
|
|
1227
|
+
evaluations=[_evaluation_wire(e) for e in receipt.evaluations],
|
|
1228
|
+
call=call,
|
|
1229
|
+
failure=failure,
|
|
1230
|
+
extra_secret_values=self._extra_secret_values,
|
|
1231
|
+
)
|
|
1232
|
+
push_result = await asyncio.to_thread(
|
|
1233
|
+
self._guarded, lambda: self._results.push(wire)
|
|
1234
|
+
)
|
|
1235
|
+
if push_result is not None and push_result.error is not None:
|
|
1236
|
+
# The contract's own obligation for a permanent rejection (e.g. 409 receipt_conflict,
|
|
1237
|
+
# 422 artifact_unknown): "the platform keeps the first; guest logs, no retry." `push()`
|
|
1238
|
+
# returns this rather than raising, so nothing inspected it before now.
|
|
1239
|
+
await self.log(
|
|
1240
|
+
level="error",
|
|
1241
|
+
message=(
|
|
1242
|
+
f"receipt for {receipt.scenario_key} rejected by the platform: "
|
|
1243
|
+
f"{push_result.error.code}: {push_result.error.message}"
|
|
1244
|
+
),
|
|
1245
|
+
)
|
|
1246
|
+
await self.aflush_events()
|
|
1247
|
+
|
|
1248
|
+
# -- artifacts (uploaded+acked BEFORE the referencing receipt) --------------------------
|
|
1249
|
+
|
|
1250
|
+
async def upload_artifact(
|
|
1251
|
+
self,
|
|
1252
|
+
data: bytes,
|
|
1253
|
+
*,
|
|
1254
|
+
kind: ob.ArtifactKind,
|
|
1255
|
+
scenario_key: str | None = None,
|
|
1256
|
+
deadline: float | None = None,
|
|
1257
|
+
) -> str | None:
|
|
1258
|
+
"""Returns the `sha256:<64-hex>` id form the wire uses (`CallSummary.transcript_artifact`,
|
|
1259
|
+
`ArtifactManifestEntry.artifact_id`) — never the bare hex `ArtifactsClient.upload` itself
|
|
1260
|
+
takes, which is a different, easy-to-mix-up shape (this module's own report notes it)."""
|
|
1261
|
+
digest = hashlib.sha256(data).hexdigest()
|
|
1262
|
+
if digest in self._uploaded_digests:
|
|
1263
|
+
return f"sha256:{digest}"
|
|
1264
|
+
if self.is_fenced:
|
|
1265
|
+
return None
|
|
1266
|
+
# guest-side level admission + budget, both BEFORE the transport is ever touched
|
|
1267
|
+
# ("the guest enforces it first").
|
|
1268
|
+
forbidden = _ARTIFACT_LEVEL_FORBIDDEN_KINDS.get(
|
|
1269
|
+
self._artifacts_policy.level, frozenset()
|
|
1270
|
+
)
|
|
1271
|
+
if kind in forbidden:
|
|
1272
|
+
await self.log(
|
|
1273
|
+
level="error",
|
|
1274
|
+
message=(
|
|
1275
|
+
f"artifact upload refused: kind={kind.value} forbidden at "
|
|
1276
|
+
f"level={self._artifacts_policy.level.value}"
|
|
1277
|
+
),
|
|
1278
|
+
)
|
|
1279
|
+
return None
|
|
1280
|
+
# check-and-reserve atomically, before the actual (slow, concurrency-safe) upload --
|
|
1281
|
+
# see the lock's own comment in __init__. A failed upload below leaves the reservation in
|
|
1282
|
+
# place rather than releasing it (`ArtifactBudgetTracker` has no release primitive): a
|
|
1283
|
+
# stuck-conservative budget is safe, an under-counted one that lets two racing uploads both
|
|
1284
|
+
# pass admission is not.
|
|
1285
|
+
async with self._artifact_budget_lock:
|
|
1286
|
+
if not self._budget_tracker.would_admit(kind, len(data), digest=digest):
|
|
1287
|
+
await self.log(
|
|
1288
|
+
level="error",
|
|
1289
|
+
message=(
|
|
1290
|
+
f"artifact upload refused: budget exhausted (kind={kind.value}, "
|
|
1291
|
+
f"size={len(data)})"
|
|
1292
|
+
),
|
|
1293
|
+
)
|
|
1294
|
+
return None
|
|
1295
|
+
self._budget_tracker.record(kind, len(data), digest=digest)
|
|
1296
|
+
result = await asyncio.to_thread(
|
|
1297
|
+
self._guarded,
|
|
1298
|
+
lambda: self._artifacts.upload(
|
|
1299
|
+
digest, data, kind=kind, scenario_key=scenario_key, deadline=deadline
|
|
1300
|
+
),
|
|
1301
|
+
)
|
|
1302
|
+
if result is None or result.error is not None:
|
|
1303
|
+
code = (
|
|
1304
|
+
result.error.code
|
|
1305
|
+
if result is not None and result.error is not None
|
|
1306
|
+
else "fenced"
|
|
1307
|
+
)
|
|
1308
|
+
await self.log(
|
|
1309
|
+
level="error", message=f"artifact upload failed ({kind.value}): {code}"
|
|
1310
|
+
)
|
|
1311
|
+
return None
|
|
1312
|
+
self._uploaded_digests.add(digest)
|
|
1313
|
+
self._manifest_entries.append(
|
|
1314
|
+
{
|
|
1315
|
+
"artifact_id": f"sha256:{digest}",
|
|
1316
|
+
"kind": kind.value,
|
|
1317
|
+
"size": len(data),
|
|
1318
|
+
"scenario_key": scenario_key,
|
|
1319
|
+
}
|
|
1320
|
+
)
|
|
1321
|
+
return f"sha256:{digest}"
|
|
1322
|
+
|
|
1323
|
+
async def push_manifest(
|
|
1324
|
+
self, *, complete: bool, deadline: float | None = None
|
|
1325
|
+
) -> bool:
|
|
1326
|
+
if self.is_fenced:
|
|
1327
|
+
return False
|
|
1328
|
+
wire = ob.build_artifact_manifest(
|
|
1329
|
+
job_id=self._capabilities.job_id,
|
|
1330
|
+
attempt_id=self._capabilities.attempt_id,
|
|
1331
|
+
attempt_number=self._capabilities.attempt_number,
|
|
1332
|
+
entries=list(self._manifest_entries),
|
|
1333
|
+
complete=complete,
|
|
1334
|
+
)
|
|
1335
|
+
result = await asyncio.to_thread(
|
|
1336
|
+
self._guarded,
|
|
1337
|
+
lambda: self._artifacts.push_manifest(wire, deadline=deadline),
|
|
1338
|
+
)
|
|
1339
|
+
if result is None:
|
|
1340
|
+
return False
|
|
1341
|
+
if not result.delivered:
|
|
1342
|
+
error = result.error
|
|
1343
|
+
logger.error(
|
|
1344
|
+
"artifact manifest delivery failed: code=%s message=%s",
|
|
1345
|
+
error.code if error is not None else "unknown",
|
|
1346
|
+
error.message if error is not None else "no response",
|
|
1347
|
+
)
|
|
1348
|
+
return False
|
|
1349
|
+
return True
|
|
1350
|
+
|
|
1351
|
+
async def ensure_terminal_artifacts(
|
|
1352
|
+
self,
|
|
1353
|
+
*,
|
|
1354
|
+
work_directory: Path,
|
|
1355
|
+
stage: HarnessStage,
|
|
1356
|
+
failure: dict[str, Any] | None,
|
|
1357
|
+
) -> None:
|
|
1358
|
+
"""Upload the three artifacts a complete platform manifest requires.
|
|
1359
|
+
|
|
1360
|
+
The platform contract requires ``build``, ``result`` and ``log`` even when a run has no
|
|
1361
|
+
transcript/recording. Previously the adapter only uploaded a log when an event was
|
|
1362
|
+
rejected, so every otherwise-successful complete manifest was deterministically rejected.
|
|
1363
|
+
"""
|
|
1364
|
+
build_path = work_directory / "build.json"
|
|
1365
|
+
build = (
|
|
1366
|
+
build_path.read_bytes()
|
|
1367
|
+
if build_path.is_file()
|
|
1368
|
+
else b'{"status":"build metadata unavailable"}\n'
|
|
1369
|
+
)
|
|
1370
|
+
result = (
|
|
1371
|
+
json.dumps(
|
|
1372
|
+
{
|
|
1373
|
+
"stage": stage.value,
|
|
1374
|
+
"failure": failure,
|
|
1375
|
+
"scenario_counts": self.scenario_counts,
|
|
1376
|
+
},
|
|
1377
|
+
sort_keys=True,
|
|
1378
|
+
separators=(",", ":"),
|
|
1379
|
+
).encode()
|
|
1380
|
+
+ b"\n"
|
|
1381
|
+
)
|
|
1382
|
+
log = (
|
|
1383
|
+
f"hosted harness terminal stage={stage.value}; "
|
|
1384
|
+
f"scenario_counts={json.dumps(self.scenario_counts, sort_keys=True)}\n"
|
|
1385
|
+
).encode()
|
|
1386
|
+
for body, kind in (
|
|
1387
|
+
(build, ob.ArtifactKind.BUILD),
|
|
1388
|
+
(result, ob.ArtifactKind.RESULT),
|
|
1389
|
+
(log, ob.ArtifactKind.LOG),
|
|
1390
|
+
):
|
|
1391
|
+
await self.upload_artifact(body, kind=kind, deadline=self.deadline())
|
|
1392
|
+
|
|
1393
|
+
# -- terminal (exactly one terminal event, last emitted) --------------------------------
|
|
1394
|
+
|
|
1395
|
+
async def emit_terminal(
|
|
1396
|
+
self,
|
|
1397
|
+
*,
|
|
1398
|
+
stage: HarnessStage,
|
|
1399
|
+
reason: ob.TerminalReason | None = None,
|
|
1400
|
+
failure: dict[str, Any] | None = None,
|
|
1401
|
+
) -> bool:
|
|
1402
|
+
"""Returns whether a terminal event was actually emitted (False when already emitted, or
|
|
1403
|
+
fenced). No caller reads this return value any more -- `drain()`'s own manifest push is
|
|
1404
|
+
gated on `is_fenced` instead; kept `bool` since a future caller may still want it."""
|
|
1405
|
+
if self._terminal_emitted or self.is_fenced:
|
|
1406
|
+
return False
|
|
1407
|
+
if failure is not None and isinstance(failure.get("message"), str):
|
|
1408
|
+
# redact BEFORE truncating -- the inverse order can cut a secret in half at the 4KB
|
|
1409
|
+
# boundary, and exact-substring redaction can no longer find the surviving fragment.
|
|
1410
|
+
redacted = ob.redact_outbound_text(
|
|
1411
|
+
failure["message"], self._extra_secret_values
|
|
1412
|
+
)
|
|
1413
|
+
failure = {**failure, "message": _cap_failure_message(redacted)}
|
|
1414
|
+
# the latch is set AFTER a successful append (below), not before -- a raise inside
|
|
1415
|
+
# `_emit_event` (an oversized payload, an invalid `failure.domain`) must not permanently
|
|
1416
|
+
# disable the terminal event.
|
|
1417
|
+
self._emit_event(
|
|
1418
|
+
stage=stage,
|
|
1419
|
+
type_=ob.OutboundEventType.TERMINAL,
|
|
1420
|
+
payload={
|
|
1421
|
+
"stage": stage.value,
|
|
1422
|
+
"reason": reason.value if reason is not None else None,
|
|
1423
|
+
"failure": failure,
|
|
1424
|
+
"scenario_counts": dict(self._scenario_counts),
|
|
1425
|
+
},
|
|
1426
|
+
)
|
|
1427
|
+
self._terminal_emitted = True
|
|
1428
|
+
# arm the flush window HERE, unconditionally -- "120s from the cancel signal / TTL /
|
|
1429
|
+
# terminal event," not only when a cancel was separately observed.
|
|
1430
|
+
self.arm_flush_window()
|
|
1431
|
+
return True
|
|
1432
|
+
|
|
1433
|
+
async def flush_terminal(self, *, deadline: float | None = None) -> bool:
|
|
1434
|
+
"""`emit_terminal` only appends the terminal record to the LOCAL spool -- a
|
|
1435
|
+
caller that pushes something else on the wire right after (skipped receipts) would
|
|
1436
|
+
otherwise risk `receipt()`'s own trailing `aflush_events()` delivering the terminal as a
|
|
1437
|
+
side effect of pushing THAT receipt, landing the terminal after it on the wire. Same
|
|
1438
|
+
bounded loop as `drain()` (a backlog bigger than one `EVENTS_MAX_BATCH` must not
|
|
1439
|
+
strand the terminal), stopping short of the manifest push -- that still belongs after
|
|
1440
|
+
skipped receipts, not here. Returns whether a fence was observed."""
|
|
1441
|
+
while True:
|
|
1442
|
+
before = self._spool.watermark()
|
|
1443
|
+
await self.aflush_events(deadline=deadline)
|
|
1444
|
+
if self.is_fenced or not self._spool.pending_since_watermark():
|
|
1445
|
+
break
|
|
1446
|
+
if self._spool.watermark() == before or (
|
|
1447
|
+
deadline is not None and time.monotonic() >= deadline
|
|
1448
|
+
):
|
|
1449
|
+
break
|
|
1450
|
+
return self.is_fenced
|
|
1451
|
+
|
|
1452
|
+
async def drain(self, *, complete: bool, deadline: float | None = None) -> bool:
|
|
1453
|
+
"""Best-effort final delivery: events, then the artifact manifest (Sequencing: "terminal
|
|
1454
|
+
event -> receipts (incl. synthesized skipped) -> manifest" -- receipts are already pushed
|
|
1455
|
+
individually by `receipt()` as each scenario finishes). Returns whether a fence was
|
|
1456
|
+
observed during (or before) this call -- the fence most often lands on the very flush
|
|
1457
|
+
that carries the terminal event, so the caller's exit code must come from THIS return
|
|
1458
|
+
value, never a fence check taken before drain() ran.
|
|
1459
|
+
|
|
1460
|
+
`aflush_events` delivers at most ONE `EVENTS_MAX_BATCH`-sized batch per call -- a
|
|
1461
|
+
backlog bigger than that (the rejected-event logging in `aflush_events` can grow one) would otherwise
|
|
1462
|
+
strand the terminal event, the highest sequence, undelivered while still exiting 0. Loops
|
|
1463
|
+
until the spool is actually empty, a fence is observed, a flush makes no further progress,
|
|
1464
|
+
or the deadline is gone -- whichever comes first.
|
|
1465
|
+
"""
|
|
1466
|
+
while True:
|
|
1467
|
+
before = self._spool.watermark()
|
|
1468
|
+
await self.aflush_events(deadline=deadline)
|
|
1469
|
+
if self.is_fenced or not self._spool.pending_since_watermark():
|
|
1470
|
+
break
|
|
1471
|
+
if self._spool.watermark() == before or (
|
|
1472
|
+
deadline is not None and time.monotonic() >= deadline
|
|
1473
|
+
):
|
|
1474
|
+
break
|
|
1475
|
+
await self.push_manifest(complete=complete, deadline=deadline)
|
|
1476
|
+
return self.is_fenced
|
|
1477
|
+
|
|
1478
|
+
|
|
1479
|
+
def _evaluation_wire(evaluation: Any) -> dict[str, Any]:
|
|
1480
|
+
if evaluation.kind == "metric":
|
|
1481
|
+
return {
|
|
1482
|
+
# `MetricEvaluation.score: float` coerces `1` -> `1.0`; `build_result_receipt`
|
|
1483
|
+
# digests the RAW dict before that coercion, so an int here would digest-mismatch
|
|
1484
|
+
# against the model's own re-derivation and silently drop the receipt.
|
|
1485
|
+
"name": evaluation.name,
|
|
1486
|
+
"kind": "metric",
|
|
1487
|
+
"score": float(evaluation.score) if evaluation.score is not None else None,
|
|
1488
|
+
"reason": evaluation.reason,
|
|
1489
|
+
}
|
|
1490
|
+
return {
|
|
1491
|
+
"name": evaluation.name,
|
|
1492
|
+
"kind": "checkpoint",
|
|
1493
|
+
"passed": evaluation.passed,
|
|
1494
|
+
"reason": evaluation.reason,
|
|
1495
|
+
}
|
|
1496
|
+
|
|
1497
|
+
|
|
1498
|
+
# =================================================================================================
|
|
1499
|
+
# Cancellation -- spine §0 step 7 / outbound-channels.md "Cancellation signal": the gateway writes
|
|
1500
|
+
# `cancel_path` then sends SIGTERM; the guest stops LAUNCHING new scenarios (not killing what's
|
|
1501
|
+
# already running) and starts the 120s flush window.
|
|
1502
|
+
# =================================================================================================
|
|
1503
|
+
|
|
1504
|
+
|
|
1505
|
+
class CancelState:
|
|
1506
|
+
def __init__(self, path: Path) -> None:
|
|
1507
|
+
self._path = path
|
|
1508
|
+
self._sigterm_seen = False
|
|
1509
|
+
|
|
1510
|
+
def note_sigterm(self) -> None:
|
|
1511
|
+
self._sigterm_seen = True
|
|
1512
|
+
|
|
1513
|
+
def requested(self) -> bool:
|
|
1514
|
+
return self._sigterm_seen or self._path.exists()
|
|
1515
|
+
|
|
1516
|
+
def reason(self) -> ob.TerminalReason | None:
|
|
1517
|
+
try:
|
|
1518
|
+
raw = json.loads(self._path.read_text(encoding="utf-8"))
|
|
1519
|
+
except (OSError, ValueError):
|
|
1520
|
+
return None
|
|
1521
|
+
value = raw.get("reason") if isinstance(raw, dict) else None
|
|
1522
|
+
try:
|
|
1523
|
+
return ob.TerminalReason(value)
|
|
1524
|
+
except ValueError:
|
|
1525
|
+
return None
|
|
1526
|
+
|
|
1527
|
+
|
|
1528
|
+
def install_sigterm_handler(cancel_state: CancelState) -> Callable[[], None]:
|
|
1529
|
+
"""Best-effort: `signal.signal` only works on the process's main thread and raises
|
|
1530
|
+
`ValueError` anywhere else (e.g. inside a test running on a worker thread) — caught and
|
|
1531
|
+
turned into a no-op restore, since `cancel_requested` still works off the file alone."""
|
|
1532
|
+
|
|
1533
|
+
def _handler(signum: int, frame: Any) -> None:
|
|
1534
|
+
del signum, frame
|
|
1535
|
+
cancel_state.note_sigterm()
|
|
1536
|
+
|
|
1537
|
+
try:
|
|
1538
|
+
previous = signal.signal(signal.SIGTERM, _handler)
|
|
1539
|
+
except (ValueError, OSError):
|
|
1540
|
+
return lambda: None
|
|
1541
|
+
|
|
1542
|
+
def _restore() -> None:
|
|
1543
|
+
try:
|
|
1544
|
+
signal.signal(signal.SIGTERM, previous)
|
|
1545
|
+
except (ValueError, OSError):
|
|
1546
|
+
pass
|
|
1547
|
+
|
|
1548
|
+
return _restore
|
|
1549
|
+
|
|
1550
|
+
|
|
1551
|
+
def default_install_sigterm_handler(cancel_state: CancelState) -> Callable[[], None]:
|
|
1552
|
+
return install_sigterm_handler(cancel_state)
|
|
1553
|
+
|
|
1554
|
+
|
|
1555
|
+
def _resolve_hosted_public_url(
|
|
1556
|
+
capabilities: ob.HostedCapabilities,
|
|
1557
|
+
transport: ob.Transport,
|
|
1558
|
+
*,
|
|
1559
|
+
port: int,
|
|
1560
|
+
expires_in_seconds: int,
|
|
1561
|
+
) -> str:
|
|
1562
|
+
endpoint = capabilities.endpoints.ingress
|
|
1563
|
+
if not endpoint:
|
|
1564
|
+
raise ProcessRuntimeError(
|
|
1565
|
+
"provider_lifecycle",
|
|
1566
|
+
"spawn_failed",
|
|
1567
|
+
"the platform did not grant an ingress capability for provider webhooks",
|
|
1568
|
+
domain=FailureDomain.INFRASTRUCTURE,
|
|
1569
|
+
)
|
|
1570
|
+
try:
|
|
1571
|
+
response = transport.request(
|
|
1572
|
+
"POST",
|
|
1573
|
+
endpoint,
|
|
1574
|
+
headers=capabilities.auth_headers(),
|
|
1575
|
+
json_body={
|
|
1576
|
+
"port": port,
|
|
1577
|
+
"expires_in_seconds": expires_in_seconds,
|
|
1578
|
+
},
|
|
1579
|
+
timeout=30.0,
|
|
1580
|
+
)
|
|
1581
|
+
except ob.TransportError as exc:
|
|
1582
|
+
raise ProcessRuntimeError(
|
|
1583
|
+
"provider_lifecycle",
|
|
1584
|
+
"spawn_failed",
|
|
1585
|
+
f"platform ingress request failed: {exc}",
|
|
1586
|
+
domain=FailureDomain.INFRASTRUCTURE,
|
|
1587
|
+
) from exc
|
|
1588
|
+
body = response.body or {}
|
|
1589
|
+
url = body.get("url")
|
|
1590
|
+
if (
|
|
1591
|
+
response.status_code != 200
|
|
1592
|
+
or not isinstance(url, str)
|
|
1593
|
+
or not url.startswith("https://")
|
|
1594
|
+
):
|
|
1595
|
+
raise ProcessRuntimeError(
|
|
1596
|
+
"provider_lifecycle",
|
|
1597
|
+
"spawn_failed",
|
|
1598
|
+
f"platform ingress request was rejected with HTTP {response.status_code}",
|
|
1599
|
+
domain=FailureDomain.INFRASTRUCTURE,
|
|
1600
|
+
)
|
|
1601
|
+
return url
|
|
1602
|
+
|
|
1603
|
+
|
|
1604
|
+
# =================================================================================================
|
|
1605
|
+
# Dependency injection -- every seam a test needs to replace with a fake, gathered in one place so
|
|
1606
|
+
# `run_job` itself stays pure orchestration.
|
|
1607
|
+
# =================================================================================================
|
|
1608
|
+
|
|
1609
|
+
|
|
1610
|
+
@dataclass
|
|
1611
|
+
class HostedEntrypointDeps:
|
|
1612
|
+
load_capabilities: Callable[[], ob.HostedCapabilities] = field(
|
|
1613
|
+
default=lambda: ob.load_capabilities()
|
|
1614
|
+
)
|
|
1615
|
+
bundle_source: BundleSource = field(default_factory=DefaultBundleSource)
|
|
1616
|
+
scenario_source: ScenarioSource = field(default_factory=NotWiredScenarioSource)
|
|
1617
|
+
build_transport: Callable[[], ob.Transport] = field(
|
|
1618
|
+
default=lambda: ob.RequestsTransport()
|
|
1619
|
+
)
|
|
1620
|
+
# Daytona forces the sandbox to a fixed non-root user (svc-control) and ignores os_user
|
|
1621
|
+
# overrides, so the guest cannot setuid/chown to the bundle's svc-agent/svc-tools/svc-data
|
|
1622
|
+
# users -- every process runs uniformly as svc-control. The bundle may still DECLARE those
|
|
1623
|
+
# users (the model validates them); they are simply not enforced at runtime here.
|
|
1624
|
+
build_provider: Callable[
|
|
1625
|
+
[ob.HostedCapabilities, ob.Transport], WorldProvisioner
|
|
1626
|
+
] = field(
|
|
1627
|
+
default=lambda capabilities, transport: ProcessRuntimeProvider(
|
|
1628
|
+
user_resolver=lambda _name: None,
|
|
1629
|
+
require_declared_user=False,
|
|
1630
|
+
public_url_resolver=lambda port, ttl: _resolve_hosted_public_url(
|
|
1631
|
+
capabilities, transport, port=port, expires_in_seconds=ttl
|
|
1632
|
+
),
|
|
1633
|
+
provider_attempt_id=capabilities.attempt_id,
|
|
1634
|
+
provider_expires_at=capabilities.expires_at,
|
|
1635
|
+
)
|
|
1636
|
+
)
|
|
1637
|
+
# The real call runner needs `OutboundAdapter.upload_artifact` to satisfy the invariant that
|
|
1638
|
+
# referenced artifacts are uploaded+acked BEFORE the receipt that names them -- the adapter is
|
|
1639
|
+
# threaded in once `run_job` has built it, rather than the CallRunner reaching for a global.
|
|
1640
|
+
# `CallRunnerContext` carries everything else `CallRunnerImpl` needs (job, bundle_dir,
|
|
1641
|
+
# evidence_seam, the target_provider secret map, attempt_number) that the bare
|
|
1642
|
+
# `CallRunner.run(scenario, runtime)` protocol has no room for.
|
|
1643
|
+
build_call_runner: Callable[["OutboundAdapter", CallRunnerContext], CallRunner] = (
|
|
1644
|
+
field(
|
|
1645
|
+
default=lambda adapter, context: _default_build_call_runner(
|
|
1646
|
+
adapter, context
|
|
1647
|
+
)
|
|
1648
|
+
)
|
|
1649
|
+
)
|
|
1650
|
+
build_world_factory: Callable[[Path], WorldFactory] = field(
|
|
1651
|
+
default=ProcessWorldFactory
|
|
1652
|
+
)
|
|
1653
|
+
retry_policy: Callable[[], ob.RetryPolicy] = field(default=lambda: ob.RetryPolicy())
|
|
1654
|
+
clock: Callable[[], datetime] = field(default=lambda: datetime.now(timezone.utc))
|
|
1655
|
+
cancel_path: Path = field(default_factory=lambda: Path(CANCEL_SIGNAL_PATH))
|
|
1656
|
+
secrets_path: Path = field(default_factory=lambda: SECRETS_PATH)
|
|
1657
|
+
simulator_secrets_path: Path = field(default_factory=lambda: SIMULATOR_SECRETS_PATH)
|
|
1658
|
+
flush_window_seconds: float = ob.FLUSH_WINDOW_SECONDS
|
|
1659
|
+
install_sigterm_handler: Callable[[CancelState], Callable[[], None]] = field(
|
|
1660
|
+
default=default_install_sigterm_handler
|
|
1661
|
+
)
|
|
1662
|
+
events_spool_dir_name: str = EVENTS_SPOOL_DIR_NAME
|
|
1663
|
+
scenarios_client_kwargs: dict[str, Any] = field(default_factory=dict)
|
|
1664
|
+
|
|
1665
|
+
def build_events_spool(self, work_directory: Path) -> ob.OutboundSpool:
|
|
1666
|
+
return ob.OutboundSpool(
|
|
1667
|
+
work_directory / self.events_spool_dir_name, "events", sequenced=True
|
|
1668
|
+
)
|
|
1669
|
+
|
|
1670
|
+
def build_scenarios_client(
|
|
1671
|
+
self,
|
|
1672
|
+
capabilities: ob.HostedCapabilities,
|
|
1673
|
+
transport: ob.Transport,
|
|
1674
|
+
channel_state: ob.ChannelState,
|
|
1675
|
+
) -> ScenariosClient:
|
|
1676
|
+
return ScenariosClient(
|
|
1677
|
+
capabilities,
|
|
1678
|
+
transport,
|
|
1679
|
+
channel_state=channel_state,
|
|
1680
|
+
**self.scenarios_client_kwargs,
|
|
1681
|
+
)
|
|
1682
|
+
|
|
1683
|
+
def peek_secret_values(self) -> tuple[str, ...]:
|
|
1684
|
+
return peek_secret_values(self.secrets_path)
|
|
1685
|
+
|
|
1686
|
+
def peek_target_provider_secret_values(
|
|
1687
|
+
self, secret_purposes: dict[str, str]
|
|
1688
|
+
) -> dict[str, str]:
|
|
1689
|
+
return peek_target_provider_secret_values(self.secrets_path, secret_purposes)
|
|
1690
|
+
|
|
1691
|
+
def peek_simulator_provider_secret_values(
|
|
1692
|
+
self, secret_purposes: dict[str, str]
|
|
1693
|
+
) -> dict[str, str]:
|
|
1694
|
+
return peek_simulator_provider_secret_values(self.secrets_path, secret_purposes)
|
|
1695
|
+
|
|
1696
|
+
def load_simulator_secret_values(self) -> dict[str, str]:
|
|
1697
|
+
return load_simulator_secret_values(self.simulator_secrets_path)
|
|
1698
|
+
|
|
1699
|
+
|
|
1700
|
+
# =================================================================================================
|
|
1701
|
+
# Scenario-entry validation at fetch (defense against karthik-integration-changes.md K1): the
|
|
1702
|
+
# Scenario Generation Contract's own model may not carry `scenario_key` (or may hand back some
|
|
1703
|
+
# other malformed shape) by the time `scenario_source.build()` returns it here, and
|
|
1704
|
+
# `hosted_scheduler.py` reads `scenario.scenario_key`/`.sub_goals`/`.setup`/`.ready` at its own
|
|
1705
|
+
# call sites with plain attribute access -- an attribute a pydantic/dataclass model never defined
|
|
1706
|
+
# raises AttributeError, not a typed failure, deep inside the scheduler with no terminal event and
|
|
1707
|
+
# a nonzero exit that reads as an infrastructure crash. Checked here with `getattr` (never direct
|
|
1708
|
+
# attribute access) so a malformed entry is caught at the seam, before the scheduler ever touches
|
|
1709
|
+
# it -- one bad entry fails the whole job as a typed FAILED terminal instead of crashing the guest.
|
|
1710
|
+
# =================================================================================================
|
|
1711
|
+
|
|
1712
|
+
# No closed-vocabulary code names this defect specifically (the §2e/§2f tables are bundle/process
|
|
1713
|
+
# concerns, not scenario-content ones) -- `scenario_preallocation_failed` is this module's own
|
|
1714
|
+
# existing code for "the scenario set is not viable for this attempt," already scoped to stage
|
|
1715
|
+
# `validating_scenarios`, and is reused here rather than inventing a new one. Domain `environment`
|
|
1716
|
+
# (not `platform_sync`, its other use here): a malformed entry is a deterministic generation-stage
|
|
1717
|
+
# content defect, not a transport failure, and fails identically on retry.
|
|
1718
|
+
_SCENARIO_ENTRY_INVALID_CODE = "scenario_preallocation_failed"
|
|
1719
|
+
|
|
1720
|
+
|
|
1721
|
+
def _validate_scenario_entry(entry: Any, *, index: int) -> str | None:
|
|
1722
|
+
"""Returns a human-readable defect description, or `None` if `entry` looks usable by
|
|
1723
|
+
`hosted_scheduler.py`'s `Scenario` Protocol. Every check is a `getattr` with a default, never
|
|
1724
|
+
a direct attribute/index access -- the whole point is to survive a shape that lacks a field
|
|
1725
|
+
entirely, not just one that carries a wrong value.
|
|
1726
|
+
"""
|
|
1727
|
+
scenario_key = getattr(entry, "scenario_key", None)
|
|
1728
|
+
if not isinstance(scenario_key, str) or not scenario_key:
|
|
1729
|
+
return f"scenario[{index}] has no non-empty scenario_key"
|
|
1730
|
+
label = f"scenario[{index}] ({scenario_key!r})"
|
|
1731
|
+
if not isinstance(getattr(entry, "scenario_id", None), str):
|
|
1732
|
+
return f"{label} has no scenario_id"
|
|
1733
|
+
if not callable(getattr(entry, "setup", None)):
|
|
1734
|
+
return f"{label} has no callable setup()"
|
|
1735
|
+
if not callable(getattr(entry, "ready", None)):
|
|
1736
|
+
return f"{label} has no callable ready()"
|
|
1737
|
+
sub_goals = getattr(entry, "sub_goals", None)
|
|
1738
|
+
if not isinstance(sub_goals, Sequence) or isinstance(sub_goals, (str, bytes)):
|
|
1739
|
+
return f"{label} has no sub_goals sequence"
|
|
1740
|
+
for goal_index, goal in enumerate(sub_goals):
|
|
1741
|
+
goal_name = getattr(goal, "name", None)
|
|
1742
|
+
if not isinstance(goal_name, str) or not goal_name:
|
|
1743
|
+
return f"{label} sub_goal[{goal_index}] has no non-empty name"
|
|
1744
|
+
if not callable(getattr(goal, "check", None)):
|
|
1745
|
+
return f"{label} sub_goal[{goal_index}] ({goal_name!r}) has no callable check()"
|
|
1746
|
+
return None
|
|
1747
|
+
|
|
1748
|
+
|
|
1749
|
+
def _validate_scenarios(scenarios: Sequence[Any]) -> str | None:
|
|
1750
|
+
for index, entry in enumerate(scenarios):
|
|
1751
|
+
defect = _validate_scenario_entry(entry, index=index)
|
|
1752
|
+
if defect is not None:
|
|
1753
|
+
return defect
|
|
1754
|
+
return None
|
|
1755
|
+
|
|
1756
|
+
|
|
1757
|
+
# =================================================================================================
|
|
1758
|
+
# Orchestration -- steps 1-8, in order.
|
|
1759
|
+
# =================================================================================================
|
|
1760
|
+
|
|
1761
|
+
|
|
1762
|
+
async def run_job(
|
|
1763
|
+
job_path: Path,
|
|
1764
|
+
source: Path,
|
|
1765
|
+
output: Path,
|
|
1766
|
+
*,
|
|
1767
|
+
deps: HostedEntrypointDeps | None = None,
|
|
1768
|
+
) -> int:
|
|
1769
|
+
"""The guest's whole `main()` body. Returns the process exit code (§0.6) — `main()` below is
|
|
1770
|
+
the only caller that turns this into `SystemExit`, so tests can call this directly and assert
|
|
1771
|
+
on the return value."""
|
|
1772
|
+
deps = deps or HostedEntrypointDeps()
|
|
1773
|
+
work_directory = output.parent
|
|
1774
|
+
|
|
1775
|
+
# This control-process-only channel is loaded before any Stage or CallRunner is constructed.
|
|
1776
|
+
# It is separate from secrets.json so platform simulator credentials never acquire the
|
|
1777
|
+
# ``target_provider`` purpose and therefore can never enter an agent process.
|
|
1778
|
+
simulator_secret_values = deps.load_simulator_secret_values()
|
|
1779
|
+
os.environ.update(simulator_secret_values)
|
|
1780
|
+
|
|
1781
|
+
# 1. Boot -- capabilities. CapabilitiesError -> exit non-zero-and-NOT-3, no event (v1.3 table):
|
|
1782
|
+
# there is no channel yet to report a terminal event through.
|
|
1783
|
+
try:
|
|
1784
|
+
capabilities = deps.load_capabilities()
|
|
1785
|
+
except ob.CapabilitiesError as exc:
|
|
1786
|
+
logger.error("capabilities load failed: %s: %s", exc.code, exc.message)
|
|
1787
|
+
return EXIT_BOOT_FAILURE
|
|
1788
|
+
|
|
1789
|
+
# Every line from here on is attributable. Done as early as the id is known, which is
|
|
1790
|
+
# immediately after capabilities load.
|
|
1791
|
+
configure_runner_logging(getattr(capabilities, "job_id", None))
|
|
1792
|
+
|
|
1793
|
+
channel_state = ob.ChannelState()
|
|
1794
|
+
transport = deps.build_transport()
|
|
1795
|
+
retry_policy = deps.retry_policy()
|
|
1796
|
+
events_spool = deps.build_events_spool(work_directory)
|
|
1797
|
+
events_client = ob.EventsClient(
|
|
1798
|
+
capabilities,
|
|
1799
|
+
events_spool,
|
|
1800
|
+
transport,
|
|
1801
|
+
retry_policy=retry_policy,
|
|
1802
|
+
channel_state=channel_state,
|
|
1803
|
+
)
|
|
1804
|
+
results_client = ob.ResultsClient(
|
|
1805
|
+
capabilities, transport, retry_policy=retry_policy, channel_state=channel_state
|
|
1806
|
+
)
|
|
1807
|
+
artifacts_client = ob.ArtifactsClient(
|
|
1808
|
+
capabilities, transport, retry_policy=retry_policy, channel_state=channel_state
|
|
1809
|
+
)
|
|
1810
|
+
scenarios_client = deps.build_scenarios_client(
|
|
1811
|
+
capabilities, transport, channel_state
|
|
1812
|
+
)
|
|
1813
|
+
|
|
1814
|
+
adapter = OutboundAdapter(
|
|
1815
|
+
capabilities,
|
|
1816
|
+
events_spool=events_spool,
|
|
1817
|
+
events_client=events_client,
|
|
1818
|
+
results_client=results_client,
|
|
1819
|
+
artifacts_client=artifacts_client,
|
|
1820
|
+
channel_state=channel_state,
|
|
1821
|
+
extra_secret_values=(
|
|
1822
|
+
*deps.peek_secret_values(),
|
|
1823
|
+
*tuple(simulator_secret_values.values()),
|
|
1824
|
+
),
|
|
1825
|
+
clock=deps.clock,
|
|
1826
|
+
flush_window_seconds=deps.flush_window_seconds,
|
|
1827
|
+
)
|
|
1828
|
+
|
|
1829
|
+
cancel_state = CancelState(deps.cancel_path)
|
|
1830
|
+
restore_sigterm = deps.install_sigterm_handler(cancel_state)
|
|
1831
|
+
# held outside the try so an exception on any path after this line still lets the
|
|
1832
|
+
# `finally` below close whatever was actually provisioned.
|
|
1833
|
+
pool: WorldPool | None = None
|
|
1834
|
+
call_runner: CallRunner | None = None
|
|
1835
|
+
|
|
1836
|
+
def cancel_requested() -> bool:
|
|
1837
|
+
requested = cancel_state.requested() or adapter.is_fenced
|
|
1838
|
+
if requested:
|
|
1839
|
+
# "120s from the cancel signal / TTL / terminal event" -- whichever comes first;
|
|
1840
|
+
# a cancel/fence observed here starts the clock even though the terminal event (which
|
|
1841
|
+
# also arms it, unconditionally) may not land until much later.
|
|
1842
|
+
adapter.arm_flush_window()
|
|
1843
|
+
return requested
|
|
1844
|
+
|
|
1845
|
+
async def _bounded_close() -> None:
|
|
1846
|
+
if pool is None:
|
|
1847
|
+
return
|
|
1848
|
+
remaining = adapter.deadline()
|
|
1849
|
+
if remaining is None:
|
|
1850
|
+
await pool.close()
|
|
1851
|
+
return
|
|
1852
|
+
timeout = max(0.0, remaining - time.monotonic())
|
|
1853
|
+
try:
|
|
1854
|
+
await asyncio.wait_for(pool.close(), timeout=timeout)
|
|
1855
|
+
except asyncio.TimeoutError:
|
|
1856
|
+
logger.warning(
|
|
1857
|
+
"pool.close() did not finish within the remaining flush window (%.1fs); "
|
|
1858
|
+
"WorldPool already latches itself closed on entry to close(), so the top-level "
|
|
1859
|
+
"finally's own pool.close() call cannot retry the teardown -- the provisioner may "
|
|
1860
|
+
"be left not fully torn down until close()'s latch ordering changes",
|
|
1861
|
+
timeout,
|
|
1862
|
+
)
|
|
1863
|
+
|
|
1864
|
+
async def _finish(
|
|
1865
|
+
stage: HarnessStage,
|
|
1866
|
+
*,
|
|
1867
|
+
reason: ob.TerminalReason | None = None,
|
|
1868
|
+
failure: dict[str, Any] | None = None,
|
|
1869
|
+
complete: bool,
|
|
1870
|
+
scheduler_result: tuple[HostedScheduler, RunResult] | None = None,
|
|
1871
|
+
) -> int:
|
|
1872
|
+
"""Terminal event -> drain -> bounded close, in that order, for every path that
|
|
1873
|
+
reaches a genuine terminal stage -- FAILED (via `_fail`; this now covers the pre-run
|
|
1874
|
+
failure branches too, not just post-run ones), CANCELED, an aborted RunResult, COMPLETED.
|
|
1875
|
+
Spending close()'s W-engine teardown time BEFORE a single terminal event is queued is
|
|
1876
|
+
exactly the inversion this ordering guards against.
|
|
1877
|
+
|
|
1878
|
+
`scheduler_result` is only ever passed by the three call sites reached AFTER
|
|
1879
|
+
`scheduler.run()` -- pre-run terminals (`_fail`, the boundary `_canceled()` checks) have
|
|
1880
|
+
no `RunResult` and pass nothing, so this stays a no-op there."""
|
|
1881
|
+
# Artifact bytes must be uploaded before the terminal-referenced complete manifest. The
|
|
1882
|
+
# terminal event itself remains before receipts and the manifest on the outbound channel.
|
|
1883
|
+
await adapter.ensure_terminal_artifacts(
|
|
1884
|
+
work_directory=work_directory,
|
|
1885
|
+
stage=stage,
|
|
1886
|
+
failure=failure,
|
|
1887
|
+
)
|
|
1888
|
+
await adapter.emit_terminal(stage=stage, reason=reason, failure=failure)
|
|
1889
|
+
# (outbound-channels.md v1.3 Sequencing): skipped receipts go out AFTER the terminal
|
|
1890
|
+
# event, never before -- placed here so no return path below can skip this call while
|
|
1891
|
+
# still delivering the terminal. A fenced result emits nothing further (the scheduler's
|
|
1892
|
+
# own no-op covers it too; checked here as well so a fenced run never even attempts it).
|
|
1893
|
+
# Best-effort like every other post-terminal emission in this module: a failure here must
|
|
1894
|
+
# not undo the terminal already spooled above or change the exit code below.
|
|
1895
|
+
if scheduler_result is not None:
|
|
1896
|
+
finished_scheduler, run_result = scheduler_result
|
|
1897
|
+
if run_result.fenced is None:
|
|
1898
|
+
try:
|
|
1899
|
+
# `emit_terminal` above only spools the terminal locally -- flushed to the
|
|
1900
|
+
# wire here, BEFORE the skipped-receipt pushes below, so a receipt's own
|
|
1901
|
+
# trailing flush can never deliver the terminal as a side effect and land it
|
|
1902
|
+
# after that receipt on the wire. Not itself wrapped in the wait_for below --
|
|
1903
|
+
# it already threads the same deadline through every retry it makes, and it
|
|
1904
|
+
# runs first, so its own delivery attempt is never the thing a timeout cuts off.
|
|
1905
|
+
await adapter.flush_terminal(deadline=adapter.deadline())
|
|
1906
|
+
if not adapter.is_fenced:
|
|
1907
|
+
# `emit_skipped_receipts`/`receipt()` have no deadline plumbing of their
|
|
1908
|
+
# own (`push()`/`aflush_events()` run with `deadline=None`) -- a
|
|
1909
|
+
# degraded-but-alive events channel can retry every skipped scenario's
|
|
1910
|
+
# receipt for the full `RetryPolicy` budget, scaling with how many
|
|
1911
|
+
# scenarios were cut short and blowing past the flush window the gateway
|
|
1912
|
+
# tears the sandbox down at. Bounded the same way `_bounded_close` bounds
|
|
1913
|
+
# `pool.close()`: past the deadline, stop trying and fall through to close.
|
|
1914
|
+
remaining = adapter.deadline()
|
|
1915
|
+
timeout = (
|
|
1916
|
+
None
|
|
1917
|
+
if remaining is None
|
|
1918
|
+
else max(0.0, remaining - time.monotonic())
|
|
1919
|
+
)
|
|
1920
|
+
try:
|
|
1921
|
+
await asyncio.wait_for(
|
|
1922
|
+
finished_scheduler.emit_skipped_receipts(run_result),
|
|
1923
|
+
timeout=timeout,
|
|
1924
|
+
)
|
|
1925
|
+
except asyncio.TimeoutError:
|
|
1926
|
+
logger.warning(
|
|
1927
|
+
"flush window exhausted before emit_skipped_receipts finished; "
|
|
1928
|
+
"remaining scenarios' receipts were not sent"
|
|
1929
|
+
)
|
|
1930
|
+
except Exception as exc: # noqa: BLE001 - post-terminal telemetry, never fatal
|
|
1931
|
+
logger.error("emit_skipped_receipts failed: %s", exc)
|
|
1932
|
+
# unlinked AFTER the terminal event, not before -- every terminal path shares the
|
|
1933
|
+
# "secrets are no longer needed past this point" rule, but an unlink failure (a read-only
|
|
1934
|
+
# or non-owned /run/futureagi) must never cost the one event that proves the job reached a
|
|
1935
|
+
# terminal state at all. missing_ok=True still no-ops on paths where the provider's own
|
|
1936
|
+
# §4.4 close() already removed the file; a genuine OSError is logged, not raised --
|
|
1937
|
+
# deleting an already-unneeded file is best-effort, not load-bearing.
|
|
1938
|
+
try:
|
|
1939
|
+
deps.secrets_path.unlink(missing_ok=True)
|
|
1940
|
+
except OSError as exc:
|
|
1941
|
+
logger.warning("secrets.json unlink failed: %s", exc)
|
|
1942
|
+
if adapter.is_fenced:
|
|
1943
|
+
await _bounded_close()
|
|
1944
|
+
return EXIT_FENCED
|
|
1945
|
+
# the exit code comes from drain()'s own post-hoc fence check (deadline computed AFTER
|
|
1946
|
+
# emit_terminal, which is what arms the flush window), never a stale pre-drain read.
|
|
1947
|
+
fenced = await adapter.drain(deadline=adapter.deadline(), complete=complete)
|
|
1948
|
+
await _bounded_close()
|
|
1949
|
+
if fenced:
|
|
1950
|
+
return EXIT_FENCED
|
|
1951
|
+
# §0.6 v1.14: the terminal was decided but the final drain could not flush it (the events
|
|
1952
|
+
# channel failed) or the platform permanently rejected the terminal item itself -- exit 0
|
|
1953
|
+
# would claim a flush that provably never happened and silently lose the run's evidence.
|
|
1954
|
+
if adapter.terminal_undelivered:
|
|
1955
|
+
return EXIT_TERMINAL_UNDELIVERED
|
|
1956
|
+
return EXIT_OK
|
|
1957
|
+
|
|
1958
|
+
async def _canceled(
|
|
1959
|
+
*,
|
|
1960
|
+
scheduler_result: tuple[HostedScheduler, RunResult] | None = None,
|
|
1961
|
+
) -> int:
|
|
1962
|
+
return await _finish(
|
|
1963
|
+
HarnessStage.CANCELED,
|
|
1964
|
+
reason=cancel_state.reason(),
|
|
1965
|
+
complete=False,
|
|
1966
|
+
scheduler_result=scheduler_result,
|
|
1967
|
+
)
|
|
1968
|
+
|
|
1969
|
+
async def _fail(
|
|
1970
|
+
*, domain: FailureDomain, fail_stage: HarnessStage, code: str, message: str
|
|
1971
|
+
) -> int:
|
|
1972
|
+
# routed through `_finish` -- terminal event first, pool close (bounded) after, for
|
|
1973
|
+
# every pre-run failure branch too, not just the post-run ones `_finish` already covered.
|
|
1974
|
+
return await _finish(
|
|
1975
|
+
HarnessStage.FAILED,
|
|
1976
|
+
failure={
|
|
1977
|
+
"domain": domain.value,
|
|
1978
|
+
"stage": fail_stage.value,
|
|
1979
|
+
"code": code,
|
|
1980
|
+
"message": message,
|
|
1981
|
+
},
|
|
1982
|
+
complete=True,
|
|
1983
|
+
)
|
|
1984
|
+
|
|
1985
|
+
try:
|
|
1986
|
+
# 1 (cont'd). job.json (§0.2).
|
|
1987
|
+
try:
|
|
1988
|
+
job = load_job(job_path)
|
|
1989
|
+
except Exception as exc: # noqa: BLE001 - a malformed job.json has no typed error to catch
|
|
1990
|
+
# EXIT_CRASHED (not a FAILED terminal) even though a channel now exists -- a
|
|
1991
|
+
# malformed job.json means `job.seed`/`job.agent`/etc are not trustworthy enough to
|
|
1992
|
+
# build a reportable failure from, and every downstream stage assumes a valid `job`.
|
|
1993
|
+
logger.error("job.json invalid: %s", exc)
|
|
1994
|
+
return EXIT_CRASHED
|
|
1995
|
+
|
|
1996
|
+
observability.begin(
|
|
1997
|
+
job.job_id, job.run_id, (job.metadata or {}).get("telemetry")
|
|
1998
|
+
)
|
|
1999
|
+
|
|
2000
|
+
if job.seed is None:
|
|
2001
|
+
logger.warning(
|
|
2002
|
+
"job.seed is null; spine §1 guarantees a concrete integer -- using 0"
|
|
2003
|
+
)
|
|
2004
|
+
job_seed = job.seed if job.seed is not None else 0
|
|
2005
|
+
parallelism = resolve_parallelism(job)
|
|
2006
|
+
secret_purposes = job_secret_purposes(job)
|
|
2007
|
+
# `ProcessRuntimeProvider` deletes `secrets.json` on its FIRST `provision()` call, inside
|
|
2008
|
+
# `pool.start()` below -- this capture must happen (and does: `job` is only just now
|
|
2009
|
+
# available, but `pool.start()` is still ~50 lines further down) strictly BEFORE that
|
|
2010
|
+
# point, same constraint `deps.peek_secret_values()` already satisfies for redaction,
|
|
2011
|
+
# above at adapter construction. Alias-preserving so `CallRunnerImpl` can pick e.g.
|
|
2012
|
+
# `LIVEKIT_API_KEY` out of the map by name.
|
|
2013
|
+
target_provider_secret_values = deps.peek_target_provider_secret_values(
|
|
2014
|
+
secret_purposes
|
|
2015
|
+
)
|
|
2016
|
+
# Platform simulator credentials arrive through simulator-secrets.json, not the
|
|
2017
|
+
# customer-controlled secrets.json. Preserve that separately loaded channel all the way
|
|
2018
|
+
# into CallRunnerContext. Any legacy simulator-purpose refs are merged first so the
|
|
2019
|
+
# platform channel wins on alias collisions and cannot be overridden by a submitted job.
|
|
2020
|
+
simulator_provider_secret_values = {
|
|
2021
|
+
**deps.peek_simulator_provider_secret_values(secret_purposes),
|
|
2022
|
+
**simulator_secret_values,
|
|
2023
|
+
}
|
|
2024
|
+
adapter.configure_artifacts(
|
|
2025
|
+
job.artifacts
|
|
2026
|
+
) # level table + budget, now that job.json is known.
|
|
2027
|
+
|
|
2028
|
+
adapter.stage_changed(HarnessStage.VALIDATING_ENVIRONMENT)
|
|
2029
|
+
await adapter.aflush_events()
|
|
2030
|
+
|
|
2031
|
+
# Bundle authoring is not this module's (see the class docstrings above) -- injected.
|
|
2032
|
+
try:
|
|
2033
|
+
manifest, bundle_dir = await asyncio.to_thread(
|
|
2034
|
+
deps.bundle_source.load,
|
|
2035
|
+
job,
|
|
2036
|
+
source=source,
|
|
2037
|
+
work_directory=work_directory,
|
|
2038
|
+
)
|
|
2039
|
+
except BundleUnavailableError as exc:
|
|
2040
|
+
return await _fail(
|
|
2041
|
+
domain=FailureDomain.ENVIRONMENT,
|
|
2042
|
+
fail_stage=HarnessStage.VALIDATING_ENVIRONMENT,
|
|
2043
|
+
code=exc.code,
|
|
2044
|
+
message=exc.message,
|
|
2045
|
+
)
|
|
2046
|
+
|
|
2047
|
+
# 2. Preflight -- BEFORE any provision (§2e). `parallelism` is the RAW requested value
|
|
2048
|
+
# (never clamped), so an out-of-1..8 W fails HERE with `parallelism_out_of_range`, per
|
|
2049
|
+
# §2e.7, rather than being silently laundered into a valid one.
|
|
2050
|
+
try:
|
|
2051
|
+
await asyncio.to_thread(
|
|
2052
|
+
preflight_bundle,
|
|
2053
|
+
bundle_dir,
|
|
2054
|
+
manifest,
|
|
2055
|
+
parallelism=parallelism,
|
|
2056
|
+
secret_refs=secret_purposes,
|
|
2057
|
+
)
|
|
2058
|
+
except PreflightError as exc:
|
|
2059
|
+
return await _fail(
|
|
2060
|
+
domain=FailureDomain.ENVIRONMENT,
|
|
2061
|
+
fail_stage=HarnessStage.VALIDATING_ENVIRONMENT,
|
|
2062
|
+
code=exc.code,
|
|
2063
|
+
message=exc.message,
|
|
2064
|
+
)
|
|
2065
|
+
|
|
2066
|
+
# cancel/fence check at the post-preflight stage boundary.
|
|
2067
|
+
if cancel_requested():
|
|
2068
|
+
return await _canceled()
|
|
2069
|
+
|
|
2070
|
+
# 4/5. Provision -- ProcessRuntimeProvider, hosted lane never passes
|
|
2071
|
+
# require_declared_user=False (the provider defaults it True on its own; the local lane's
|
|
2072
|
+
# opt-out is a construction-site concern, not this module's). §4.5b's provider mutex is
|
|
2073
|
+
# `WorldPool`'s own `_provider_lock` now (mutation-verified: it serializes
|
|
2074
|
+
# provision/reset/close/healthy under one lock) -- wired directly, no extra wrapper.
|
|
2075
|
+
provider = deps.build_provider(capabilities, transport)
|
|
2076
|
+
pool = WorldPool(
|
|
2077
|
+
provider,
|
|
2078
|
+
bundle=manifest,
|
|
2079
|
+
source=source,
|
|
2080
|
+
bundle_dir=bundle_dir,
|
|
2081
|
+
work_directory=work_directory,
|
|
2082
|
+
instances=parallelism,
|
|
2083
|
+
outbound=adapter,
|
|
2084
|
+
)
|
|
2085
|
+
try:
|
|
2086
|
+
await pool.start()
|
|
2087
|
+
except (ob.HostedFencedError, ob.HostedAttemptSupersededError):
|
|
2088
|
+
# defensive -- nothing today routes a channel error through `pool.start()`, but a
|
|
2089
|
+
# fenced attempt must never fall into the bare `Exception` handler below and get a
|
|
2090
|
+
# terminal FAILED event synthesized for it.
|
|
2091
|
+
await _bounded_close()
|
|
2092
|
+
return EXIT_FENCED
|
|
2093
|
+
except ob.HostedChannelFailedError as exc:
|
|
2094
|
+
# `_fail` closes the pool itself now, AFTER the terminal event (via `_finish`) --
|
|
2095
|
+
# closing here first was the same close-before-terminal inversion that the terminal -> drain -> close ordering fixes elsewhere.
|
|
2096
|
+
return await _fail(
|
|
2097
|
+
domain=FailureDomain.PLATFORM_SYNC,
|
|
2098
|
+
fail_stage=HarnessStage.VALIDATING_SCENARIOS,
|
|
2099
|
+
code="scenario_preallocation_failed",
|
|
2100
|
+
message=str(exc),
|
|
2101
|
+
)
|
|
2102
|
+
except ProcessRuntimeError as exc:
|
|
2103
|
+
# §2f's own CARRIED domain (never the flattened `infrastructure`/"provision_failed"
|
|
2104
|
+
# every provisioning failure used to get), stage `building_environment` per §2f.
|
|
2105
|
+
if adapter.is_fenced:
|
|
2106
|
+
await _bounded_close()
|
|
2107
|
+
return EXIT_FENCED
|
|
2108
|
+
return await _fail(
|
|
2109
|
+
domain=_process_runtime_error_domain(exc),
|
|
2110
|
+
fail_stage=HarnessStage.BUILDING_ENVIRONMENT,
|
|
2111
|
+
code=_section_2f_code(exc.code),
|
|
2112
|
+
message=str(exc),
|
|
2113
|
+
)
|
|
2114
|
+
except Exception as exc: # noqa: BLE001 - genuinely untyped -> infrastructure is the honest default
|
|
2115
|
+
if adapter.is_fenced:
|
|
2116
|
+
await _bounded_close()
|
|
2117
|
+
return EXIT_FENCED
|
|
2118
|
+
return await _fail(
|
|
2119
|
+
domain=FailureDomain.INFRASTRUCTURE,
|
|
2120
|
+
fail_stage=HarnessStage.BUILDING_ENVIRONMENT,
|
|
2121
|
+
code=_section_2f_code("provision_failed"),
|
|
2122
|
+
message=f"provision_failed: {exc}",
|
|
2123
|
+
)
|
|
2124
|
+
|
|
2125
|
+
# baseline_frozen + parallelism_degraded from build.json. The whole
|
|
2126
|
+
# block is guarded -- a malformed build.json value must degrade to a `log`, never kill a
|
|
2127
|
+
# run that has already provisioned real worlds.
|
|
2128
|
+
degrade_emitted = False
|
|
2129
|
+
try:
|
|
2130
|
+
build_output = await asyncio.to_thread(load_build_output, work_directory)
|
|
2131
|
+
except WorldFactoryError:
|
|
2132
|
+
build_output = {}
|
|
2133
|
+
try:
|
|
2134
|
+
for store in build_output.get("stores", []):
|
|
2135
|
+
if store.get("baseline_reference"):
|
|
2136
|
+
adapter.baseline_frozen(
|
|
2137
|
+
inputs_digest=str(store.get("inputs_digest", "")),
|
|
2138
|
+
baseline_ref=str(store.get("baseline_reference", "")),
|
|
2139
|
+
)
|
|
2140
|
+
degrade_reason = build_output.get("degrade_reason")
|
|
2141
|
+
if degrade_reason:
|
|
2142
|
+
requested = int(
|
|
2143
|
+
build_output.get("requested_parallelism") or parallelism
|
|
2144
|
+
)
|
|
2145
|
+
effective = int(build_output.get("effective_parallelism") or 1)
|
|
2146
|
+
# `ParallelismDegradedPayload` requires `1 <= effective < requested` --
|
|
2147
|
+
# `fixed_port` is recorded at `instances == 1` too (provider-side gap), where
|
|
2148
|
+
# `effective == requested == 1` is not representable as a degrade at all.
|
|
2149
|
+
if effective < requested:
|
|
2150
|
+
adapter.parallelism_degraded(
|
|
2151
|
+
requested=requested,
|
|
2152
|
+
effective=effective,
|
|
2153
|
+
reason=str(degrade_reason),
|
|
2154
|
+
)
|
|
2155
|
+
degrade_emitted = True
|
|
2156
|
+
else:
|
|
2157
|
+
await adapter.log(
|
|
2158
|
+
level="warning",
|
|
2159
|
+
message=(
|
|
2160
|
+
f"degrade recorded ({degrade_reason}) with requested==effective=="
|
|
2161
|
+
f"{requested}; no parallelism_degraded event is representable"
|
|
2162
|
+
),
|
|
2163
|
+
)
|
|
2164
|
+
except Exception as exc: # noqa: BLE001 - malformed build.json must never crash a live run
|
|
2165
|
+
await adapter.log(
|
|
2166
|
+
level="warning",
|
|
2167
|
+
message=f"build.json degrade/baseline block malformed: {exc}",
|
|
2168
|
+
)
|
|
2169
|
+
# `pool.effective_size` is the ground truth for how many worlds actually exist --
|
|
2170
|
+
# if it's short of what was requested and build.json's own `degrade_reason` didn't already
|
|
2171
|
+
# announce it (a runtime degrade build.json doesn't record), say so loudly rather
|
|
2172
|
+
# than silently.
|
|
2173
|
+
if not degrade_emitted and pool.effective_size < parallelism:
|
|
2174
|
+
await adapter.log(
|
|
2175
|
+
level="warning",
|
|
2176
|
+
message=(
|
|
2177
|
+
f"world pool effective_size={pool.effective_size} < requested "
|
|
2178
|
+
f"parallelism={parallelism}, but build.json recorded no representable "
|
|
2179
|
+
"degrade_reason"
|
|
2180
|
+
),
|
|
2181
|
+
)
|
|
2182
|
+
await adapter.aflush_events()
|
|
2183
|
+
|
|
2184
|
+
# cancel/fence check at the post-provision stage boundary.
|
|
2185
|
+
if cancel_requested():
|
|
2186
|
+
return await _canceled()
|
|
2187
|
+
|
|
2188
|
+
# 3. Scenario pre-allocation (spine §5 step 3.5). Generation is not this module's; the
|
|
2189
|
+
# pre-allocation CLIENT (ScenariosClient) is.
|
|
2190
|
+
adapter.stage_changed(HarnessStage.VALIDATING_SCENARIOS)
|
|
2191
|
+
await adapter.aflush_events()
|
|
2192
|
+
world_factory = deps.build_world_factory(work_directory)
|
|
2193
|
+
# An injected `ScenarioSource` (every test, every future caller) always wins -- the real
|
|
2194
|
+
# bundle-reading adapter (scenario_source.py) only steps in when the default
|
|
2195
|
+
# `NotWiredScenarioSource` is still in place AND the bundle actually carries a `scenarios/`
|
|
2196
|
+
# directory (the LAYOUT DECISION's presence test). A bundle without one keeps the existing
|
|
2197
|
+
# typed `ScenarioSourceNotWired` failure below -- no regression for a job whose scenarios
|
|
2198
|
+
# are not generated yet.
|
|
2199
|
+
scenario_source = deps.scenario_source
|
|
2200
|
+
if isinstance(scenario_source, NotWiredScenarioSource) and bundle_has_scenarios(
|
|
2201
|
+
bundle_dir
|
|
2202
|
+
):
|
|
2203
|
+
scenario_source = BundleScenarioSource()
|
|
2204
|
+
try:
|
|
2205
|
+
scenarios = await scenario_source.build(
|
|
2206
|
+
job,
|
|
2207
|
+
manifest,
|
|
2208
|
+
scenarios_client,
|
|
2209
|
+
pool=pool,
|
|
2210
|
+
world_factory=world_factory,
|
|
2211
|
+
bundle_dir=bundle_dir,
|
|
2212
|
+
)
|
|
2213
|
+
except (ob.HostedFencedError, ob.HostedAttemptSupersededError):
|
|
2214
|
+
# `ScenariosClient._post` re-raises these after latching `channel_state` -- a fence
|
|
2215
|
+
# here must exit 3 with no terminal event, never fall through to the generic handler.
|
|
2216
|
+
await _bounded_close()
|
|
2217
|
+
return EXIT_FENCED
|
|
2218
|
+
except ob.HostedChannelFailedError as exc:
|
|
2219
|
+
# `ScenariosClient._post` has already latched `channel_state` by the time this branch
|
|
2220
|
+
# runs -- `emit_terminal` still spools the terminal locally (it never touches the
|
|
2221
|
+
# network), but the drain that would flush it inherits the same latched channel and can
|
|
2222
|
+
# never deliver. `_finish` detects exactly this (the terminal's own spool sequence never
|
|
2223
|
+
# gets acked) and reports it honestly rather than claiming a flush that cannot happen.
|
|
2224
|
+
return await _fail(
|
|
2225
|
+
domain=FailureDomain.PLATFORM_SYNC,
|
|
2226
|
+
fail_stage=HarnessStage.VALIDATING_SCENARIOS,
|
|
2227
|
+
code="scenario_preallocation_failed",
|
|
2228
|
+
message=str(exc),
|
|
2229
|
+
)
|
|
2230
|
+
except (ScenarioSourceNotWired, ScenarioPreallocationError) as exc:
|
|
2231
|
+
if adapter.is_fenced:
|
|
2232
|
+
await _bounded_close()
|
|
2233
|
+
return EXIT_FENCED
|
|
2234
|
+
return await _fail(
|
|
2235
|
+
domain=FailureDomain.PLATFORM_SYNC,
|
|
2236
|
+
fail_stage=HarnessStage.VALIDATING_SCENARIOS,
|
|
2237
|
+
code="scenario_preallocation_failed",
|
|
2238
|
+
message=str(exc),
|
|
2239
|
+
)
|
|
2240
|
+
except ScenarioDocumentInvalid as exc:
|
|
2241
|
+
# A scenario document that will not even compile is a generation-stage content defect
|
|
2242
|
+
# (deterministic on retry), never a transport failure -- same rationale as
|
|
2243
|
+
# `_SCENARIO_ENTRY_INVALID_CODE`'s other use below, reused rather than inventing a new
|
|
2244
|
+
# code for the same pair of (domain, stage).
|
|
2245
|
+
if adapter.is_fenced:
|
|
2246
|
+
await _bounded_close()
|
|
2247
|
+
return EXIT_FENCED
|
|
2248
|
+
return await _fail(
|
|
2249
|
+
domain=FailureDomain.ENVIRONMENT,
|
|
2250
|
+
fail_stage=HarnessStage.VALIDATING_SCENARIOS,
|
|
2251
|
+
code=_SCENARIO_ENTRY_INVALID_CODE,
|
|
2252
|
+
message=str(exc),
|
|
2253
|
+
)
|
|
2254
|
+
|
|
2255
|
+
# Defense against a malformed scenario entry (K1) reaching the scheduler, which reads
|
|
2256
|
+
# `scenario_key`/`sub_goals`/`setup`/`ready` with plain attribute access and would raise
|
|
2257
|
+
# AttributeError instead of failing the job cleanly.
|
|
2258
|
+
scenario_defect = _validate_scenarios(scenarios)
|
|
2259
|
+
if scenario_defect is not None:
|
|
2260
|
+
if adapter.is_fenced:
|
|
2261
|
+
await _bounded_close()
|
|
2262
|
+
return EXIT_FENCED
|
|
2263
|
+
return await _fail(
|
|
2264
|
+
domain=FailureDomain.ENVIRONMENT,
|
|
2265
|
+
fail_stage=HarnessStage.VALIDATING_SCENARIOS,
|
|
2266
|
+
code=_SCENARIO_ENTRY_INVALID_CODE,
|
|
2267
|
+
message=scenario_defect,
|
|
2268
|
+
)
|
|
2269
|
+
|
|
2270
|
+
# cancel/fence check at the post-pre-allocation stage boundary.
|
|
2271
|
+
if cancel_requested():
|
|
2272
|
+
return await _canceled()
|
|
2273
|
+
|
|
2274
|
+
# 5/6. Scheduler wiring.
|
|
2275
|
+
adapter.stage_changed(HarnessStage.RUNNING)
|
|
2276
|
+
await adapter.aflush_events()
|
|
2277
|
+
call_runner_context = CallRunnerContext(
|
|
2278
|
+
job=job,
|
|
2279
|
+
bundle_dir=bundle_dir,
|
|
2280
|
+
work_directory=work_directory,
|
|
2281
|
+
evidence_seam=manifest.runtime.evidence_seam,
|
|
2282
|
+
target_provider_secret_values=target_provider_secret_values,
|
|
2283
|
+
simulator_provider_secret_values=simulator_provider_secret_values,
|
|
2284
|
+
attempt_number=capabilities.attempt_number,
|
|
2285
|
+
source_directory=source,
|
|
2286
|
+
)
|
|
2287
|
+
call_runner = deps.build_call_runner(adapter, call_runner_context)
|
|
2288
|
+
scheduler = HostedScheduler(
|
|
2289
|
+
pool=pool,
|
|
2290
|
+
world_factory=world_factory,
|
|
2291
|
+
call_runner=call_runner,
|
|
2292
|
+
outbound=adapter,
|
|
2293
|
+
job_seed=job_seed,
|
|
2294
|
+
cancel_requested=cancel_requested,
|
|
2295
|
+
)
|
|
2296
|
+
result: RunResult = await scheduler.run(scenarios)
|
|
2297
|
+
|
|
2298
|
+
# 7. Terminal + exit codes. Terminal -> drain -> close (bounded), never close() first.
|
|
2299
|
+
if cancel_state.requested():
|
|
2300
|
+
return await _canceled(scheduler_result=(scheduler, result))
|
|
2301
|
+
if result.aborted is not None:
|
|
2302
|
+
# v1.14 §5.4 pass-through: `result.aborted.domain`/`.code` carry straight through, not
|
|
2303
|
+
# flattened to a fixed infrastructure/world_pool_exhausted pair -- the scheduler already
|
|
2304
|
+
# resolves whether every world failed on the SAME never-retried §2f code (environment
|
|
2305
|
+
# or agent domain) or a mixed set (`world_pool_exhausted`/`infrastructure`), and this
|
|
2306
|
+
# just reports that verdict unchanged.
|
|
2307
|
+
return await _finish(
|
|
2308
|
+
HarnessStage.FAILED,
|
|
2309
|
+
failure={
|
|
2310
|
+
"domain": result.aborted.domain,
|
|
2311
|
+
"stage": HarnessStage.RUNNING.value,
|
|
2312
|
+
"code": result.aborted.code,
|
|
2313
|
+
"message": result.aborted.message,
|
|
2314
|
+
},
|
|
2315
|
+
# The scheduler has emitted the errored receipt plus synthesized skipped
|
|
2316
|
+
# receipts for every scenario before reaching this branch. "complete" is an
|
|
2317
|
+
# evidence-delivery property, not a success flag: only cancellation may submit
|
|
2318
|
+
# an intentionally partial manifest.
|
|
2319
|
+
complete=True,
|
|
2320
|
+
scheduler_result=(scheduler, result),
|
|
2321
|
+
)
|
|
2322
|
+
# `complete: true` only on a genuine, nothing-cut-short COMPLETED terminal.
|
|
2323
|
+
return await _finish(
|
|
2324
|
+
HarnessStage.COMPLETED, complete=True, scheduler_result=(scheduler, result)
|
|
2325
|
+
)
|
|
2326
|
+
finally:
|
|
2327
|
+
if pool is not None:
|
|
2328
|
+
try:
|
|
2329
|
+
await (
|
|
2330
|
+
pool.close()
|
|
2331
|
+
) # idempotent backstop for any path above that missed one.
|
|
2332
|
+
except Exception: # noqa: BLE001 - a finally must never mask the real exit path
|
|
2333
|
+
logger.exception("pool.close() failed in the run_job finally backstop")
|
|
2334
|
+
if call_runner is not None:
|
|
2335
|
+
close_call_runner = getattr(call_runner, "close", None)
|
|
2336
|
+
if callable(close_call_runner):
|
|
2337
|
+
try:
|
|
2338
|
+
result = close_call_runner()
|
|
2339
|
+
if hasattr(result, "__await__"):
|
|
2340
|
+
await result
|
|
2341
|
+
except Exception: # noqa: BLE001 - cleanup must never mask the real exit path
|
|
2342
|
+
logger.exception("call runner close failed in the run_job finally backstop")
|
|
2343
|
+
restore_sigterm()
|
|
2344
|
+
|
|
2345
|
+
|
|
2346
|
+
# =================================================================================================
|
|
2347
|
+
# CLI -- spine §0 step 5's frozen invocation line:
|
|
2348
|
+
# `python -m fi.alk.harness.hosted_entrypoint /work/job.json --source /work/source --output
|
|
2349
|
+
# /work/artifacts`
|
|
2350
|
+
# =================================================================================================
|
|
2351
|
+
|
|
2352
|
+
|
|
2353
|
+
def main(argv: list[str] | None = None) -> int:
|
|
2354
|
+
parser = argparse.ArgumentParser(prog="alk-harness-worker")
|
|
2355
|
+
parser.add_argument("job", type=Path, help="typed HarnessJob JSON (/work/job.json)")
|
|
2356
|
+
parser.add_argument(
|
|
2357
|
+
"--source", required=True, type=Path, help="/work/source checkout root"
|
|
2358
|
+
)
|
|
2359
|
+
parser.add_argument("--output", required=True, type=Path, help="/work/artifacts")
|
|
2360
|
+
args = parser.parse_args(argv)
|
|
2361
|
+
try:
|
|
2362
|
+
return asyncio.run(run_job(args.job, args.source, args.output))
|
|
2363
|
+
finally:
|
|
2364
|
+
observability.end()
|
|
2365
|
+
|
|
2366
|
+
|
|
2367
|
+
if __name__ == "__main__":
|
|
2368
|
+
raise SystemExit(main())
|
|
2369
|
+
|
|
2370
|
+
|
|
2371
|
+
__all__ = [
|
|
2372
|
+
"CANCEL_SIGNAL_PATH",
|
|
2373
|
+
"EXIT_BOOT_FAILURE",
|
|
2374
|
+
"EXIT_CRASHED",
|
|
2375
|
+
"EXIT_FENCED",
|
|
2376
|
+
"EXIT_OK",
|
|
2377
|
+
"EXIT_TERMINAL_UNDELIVERED",
|
|
2378
|
+
"BundleSource",
|
|
2379
|
+
"BundleUnavailableError",
|
|
2380
|
+
"CallRunnerNotWired",
|
|
2381
|
+
"CancelState",
|
|
2382
|
+
"DefaultBundleSource",
|
|
2383
|
+
"HostedEntrypointDeps",
|
|
2384
|
+
"NotWiredCallRunner",
|
|
2385
|
+
"NotWiredScenarioSource",
|
|
2386
|
+
"OutboundAdapter",
|
|
2387
|
+
"ProcessWorldFactory",
|
|
2388
|
+
"ScenarioPreallocationError",
|
|
2389
|
+
"ScenarioSource",
|
|
2390
|
+
"ScenarioSourceNotWired",
|
|
2391
|
+
"ScenariosClient",
|
|
2392
|
+
"WorldFactoryError",
|
|
2393
|
+
"install_sigterm_handler",
|
|
2394
|
+
"job_secret_purposes",
|
|
2395
|
+
"load_build_output",
|
|
2396
|
+
"load_job",
|
|
2397
|
+
"main",
|
|
2398
|
+
"peek_secret_values",
|
|
2399
|
+
"resolve_parallelism",
|
|
2400
|
+
"row_counts_for_capability",
|
|
2401
|
+
"run_job",
|
|
2402
|
+
]
|