agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,2218 @@
|
|
|
1
|
+
"""World pool + scenario loop — `hosted-execution-seams.md` v1.12 §4/§5, `world-handle-interface.md`
|
|
2
|
+
v3.4, `outbound-channels.md` v1.3.
|
|
3
|
+
|
|
4
|
+
Owns: leasing/releasing the W worlds `process_runtime.ProcessRuntimeProvider.provision()` hands
|
|
5
|
+
back, resetting a world to pristine before each scenario (spine §4.2), running one scenario's
|
|
6
|
+
`setup`/`ready`/checks against the world handle (the return-convention + errored-receipt table in
|
|
7
|
+
`world-handle-interface.md`), the fixed one-retry-on-a-fresh-world rule (spine §5 step 4), and
|
|
8
|
+
synthesizing a complete receipt ledger (one per scenario, `skipped` for anything never attempted).
|
|
9
|
+
|
|
10
|
+
Decoupling, deliberate:
|
|
11
|
+
- `process_runtime.py` is the real, settled provisioner — imported directly (`EnvironmentRuntime`,
|
|
12
|
+
`RuntimeState`). `WorldProvisioner` below is a structural `Protocol` matching
|
|
13
|
+
`ProcessRuntimeProvider`'s actual async shape so tests can inject a fake without touching a real
|
|
14
|
+
filesystem/subprocess tree.
|
|
15
|
+
- `OutboundPort` is this module's own minimal sink for the events/receipts it produces, typed
|
|
16
|
+
against `outbound-channels.md`'s closed vocabulary; whoever wires the real client adapts to it.
|
|
17
|
+
`outbound.py` is now committed and quiescent, so this module imports exactly three of its
|
|
18
|
+
exception types — `HostedFencedError`/`HostedChannelFailedError`/`HostedAttemptSupersededError`,
|
|
19
|
+
the full `ChannelState` "stop emitting" latch for one attempt — to recognize the one outbound
|
|
20
|
+
failure class that is NOT best-effort (a 401/403 fence, an exhausted channel, or a superseded
|
|
21
|
+
attempt must stop the run, not be logged and forgotten); nothing else from that module is
|
|
22
|
+
imported here.
|
|
23
|
+
- The Scenario Generation Contract (in review) is not available here either, so `Scenario`
|
|
24
|
+
is this module's own minimal Protocol for what the loop needs: a key/id pair, `setup`/`ready`,
|
|
25
|
+
and named sub-goal checks. Same for the simulated "call" itself (a different track's seam) —
|
|
26
|
+
`CallRunner` is injected.
|
|
27
|
+
- Secrets and the cancel signal are entrypoint-owned (P10); `cancel_requested` is an injected
|
|
28
|
+
zero-argument callable.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import asyncio
|
|
34
|
+
import inspect
|
|
35
|
+
import logging
|
|
36
|
+
import random
|
|
37
|
+
import re
|
|
38
|
+
import threading
|
|
39
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
40
|
+
from dataclasses import dataclass
|
|
41
|
+
from pathlib import Path
|
|
42
|
+
from typing import Any, Awaitable, Callable, Protocol, Sequence
|
|
43
|
+
|
|
44
|
+
from . import observability
|
|
45
|
+
from .job import FailureDomain, HarnessStage
|
|
46
|
+
from .judge import judge as _judge
|
|
47
|
+
from .outbound import (
|
|
48
|
+
HostedAttemptSupersededError,
|
|
49
|
+
HostedChannelFailedError,
|
|
50
|
+
HostedFencedError,
|
|
51
|
+
)
|
|
52
|
+
from .process_runtime import (
|
|
53
|
+
SECTION_2F_DOMAIN,
|
|
54
|
+
EnvironmentRuntime,
|
|
55
|
+
ProcessRuntimeError,
|
|
56
|
+
RuntimeState,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
from .world.errors import (
|
|
60
|
+
WorldError,
|
|
61
|
+
WorldQueryRejected,
|
|
62
|
+
WorldReadOnly,
|
|
63
|
+
WorldReservedName,
|
|
64
|
+
WorldStateTooLarge,
|
|
65
|
+
WorldUnavailable,
|
|
66
|
+
WorldUsageError,
|
|
67
|
+
)
|
|
68
|
+
from .world.runtime import Call
|
|
69
|
+
|
|
70
|
+
logger = logging.getLogger(__name__)
|
|
71
|
+
|
|
72
|
+
# --- the World handle (world-handle-interface.md v3.4) --------------------------------------
|
|
73
|
+
#
|
|
74
|
+
# The frozen contract's code block gives six verbs plus `world_index`/`rng`. `read_only()` is not
|
|
75
|
+
# in that block, but the contract still requires `ready`/`check` to receive a handle whose writes
|
|
76
|
+
# raise `WorldReadOnly` (own section, "Read-only handles") without saying how a caller gets one —
|
|
77
|
+
# the shipped `HostedWorld.read_only()` (`world/handle.py`) already names this exact operation, so
|
|
78
|
+
# mirroring it here is the reversible choice: a real `HostedWorld` satisfies this Protocol as-is.
|
|
79
|
+
#
|
|
80
|
+
# m8: `World.read_only()` used to be typed `-> "World"`, but the real `ReadOnlyWorld` it returns
|
|
81
|
+
# has no `read_only()` of its own (mirroring `world/handle.py`'s own `ReadOnlyWorld`, which is
|
|
82
|
+
# deliberately not re-enterable) — so it fails a structural check against `World` itself.
|
|
83
|
+
# `ReadOnlyWorld` below names the narrower surface `ready`/`check` actually receive.
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class ReadOnlyWorld(Protocol):
|
|
87
|
+
world_index: int
|
|
88
|
+
rng: random.Random
|
|
89
|
+
|
|
90
|
+
def state(self, table: str | None = None) -> dict[str, list[dict[str, Any]]]: ...
|
|
91
|
+
|
|
92
|
+
def put(
|
|
93
|
+
self, collection: str, record: dict[str, Any], *, key: str = ""
|
|
94
|
+
) -> dict[str, Any]: ...
|
|
95
|
+
|
|
96
|
+
def change(
|
|
97
|
+
self, collection: str, key: str, changes: dict[str, Any], *, by: str = ""
|
|
98
|
+
) -> int: ...
|
|
99
|
+
|
|
100
|
+
def drop(self, collection: str, key: str = "", *, by: str = "") -> int: ...
|
|
101
|
+
|
|
102
|
+
def call(self, name: str, arguments: dict[str, Any] | None = None) -> Call: ...
|
|
103
|
+
|
|
104
|
+
def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]: ...
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class World(Protocol):
|
|
108
|
+
world_index: int
|
|
109
|
+
rng: random.Random
|
|
110
|
+
|
|
111
|
+
def state(self, table: str | None = None) -> dict[str, list[dict[str, Any]]]: ...
|
|
112
|
+
|
|
113
|
+
def put(
|
|
114
|
+
self, collection: str, record: dict[str, Any], *, key: str = ""
|
|
115
|
+
) -> dict[str, Any]: ...
|
|
116
|
+
|
|
117
|
+
def change(
|
|
118
|
+
self, collection: str, key: str, changes: dict[str, Any], *, by: str = ""
|
|
119
|
+
) -> int: ...
|
|
120
|
+
|
|
121
|
+
def drop(self, collection: str, key: str = "", *, by: str = "") -> int: ...
|
|
122
|
+
|
|
123
|
+
def call(self, name: str, arguments: dict[str, Any] | None = None) -> Call: ...
|
|
124
|
+
|
|
125
|
+
def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]: ...
|
|
126
|
+
|
|
127
|
+
def read_only(self) -> ReadOnlyWorld: ...
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class WorldFactory(Protocol):
|
|
131
|
+
"""Builds the `World` handle for one already-reset `EnvironmentRuntime`.
|
|
132
|
+
|
|
133
|
+
Deliberately not this module's job: `HostedWorld` needs a `PostgresStore` (parsed from the
|
|
134
|
+
runtime's `database` endpoint) plus the baseline row counts the provisioner measured at
|
|
135
|
+
freeze time — both live behind `ProcessRuntimeProvider`'s private state, which §4's
|
|
136
|
+
`RuntimeProvider` Protocol never exposes. Injected instead of guessed.
|
|
137
|
+
"""
|
|
138
|
+
|
|
139
|
+
async def create(
|
|
140
|
+
self, runtime: EnvironmentRuntime, *, rng: random.Random
|
|
141
|
+
) -> World: ...
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
# --- the provisioner surface this module actually drives -------------------------------------
|
|
145
|
+
#
|
|
146
|
+
# Matches `ProcessRuntimeProvider`'s real async shape (process_runtime.py) structurally, not the
|
|
147
|
+
# older single-runtime `runtime.RuntimeProvider`. `bundle` stays `Any` — this module never reads a
|
|
148
|
+
# bundle field itself, only threads it back into `provision()` reconcile calls, so it does not
|
|
149
|
+
# need `EnvironmentBundleV2`'s own in-flux-adjacent type.
|
|
150
|
+
#
|
|
151
|
+
# M1 (spine v1.12 §4): `bundle_dir` is a required keyword — §2c seed/migration paths resolve
|
|
152
|
+
# against the verified bundle root, never against `source`. `require_declared_user` is dropped
|
|
153
|
+
# entirely: it is not in §4's signature, and the real provider now defaults it `True` on its own
|
|
154
|
+
# (the local lane opts out at provider construction, not per call).
|
|
155
|
+
# M2 (spine §4 point 3): `healthy` — declared readiness probes, not "process is running" — is a
|
|
156
|
+
# port method, not an optionally-injected callable.
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class WorldProvisioner(Protocol):
|
|
160
|
+
async def provision(
|
|
161
|
+
self,
|
|
162
|
+
bundle: Any,
|
|
163
|
+
*,
|
|
164
|
+
source: Path,
|
|
165
|
+
bundle_dir: Path,
|
|
166
|
+
work_directory: Path,
|
|
167
|
+
contract: Any | None = None,
|
|
168
|
+
instances: int = 1,
|
|
169
|
+
) -> list[EnvironmentRuntime]: ...
|
|
170
|
+
|
|
171
|
+
async def reset(
|
|
172
|
+
self, runtime: EnvironmentRuntime, *, work_directory: Path
|
|
173
|
+
) -> None: ...
|
|
174
|
+
|
|
175
|
+
async def healthy(
|
|
176
|
+
self, runtime: EnvironmentRuntime, *, work_directory: Path
|
|
177
|
+
) -> bool: ...
|
|
178
|
+
|
|
179
|
+
async def close(self, *, work_directory: Path) -> None: ...
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
# --- scenarios (this module's own minimal surface; that contract is not wired yet) -------
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class SubGoal(Protocol):
|
|
186
|
+
name: str
|
|
187
|
+
judged: (
|
|
188
|
+
str # `sub_goals[].judged` per outbound-channels.md: boolean is `judged != ""`.
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
def check(self, world: ReadOnlyWorld, calls: Sequence[Call]) -> object: ...
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
JudgeFn = Callable[..., Awaitable[tuple[bool | None, str]]]
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class Scenario(Protocol):
|
|
198
|
+
scenario_key: str
|
|
199
|
+
scenario_id: (
|
|
200
|
+
str # platform id from pre-allocation (outbound-channels.md Channel 2 "Join").
|
|
201
|
+
)
|
|
202
|
+
sub_goals: Sequence[SubGoal]
|
|
203
|
+
# True only when the reference solution contains an environment/tool action. Pure
|
|
204
|
+
# conversation scenarios (for example, refusing an unsafe request) legitimately produce no
|
|
205
|
+
# world calls and must still reach their judged checks.
|
|
206
|
+
requires_tool_evidence: bool
|
|
207
|
+
|
|
208
|
+
def setup(self, world: World) -> object: ...
|
|
209
|
+
|
|
210
|
+
def ready(self, world: ReadOnlyWorld) -> object: ...
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
# --- the simulated call (a different track's seam; injected, never built here) ---------------
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
@dataclass(frozen=True)
|
|
217
|
+
class CallOutcome:
|
|
218
|
+
calls: tuple[Call, ...]
|
|
219
|
+
turns: int
|
|
220
|
+
started_at: str | None
|
|
221
|
+
ended_at: str | None
|
|
222
|
+
duration_ms: int
|
|
223
|
+
transcript_artifact: str | None = None
|
|
224
|
+
recording_artifacts: tuple[str, ...] = ()
|
|
225
|
+
stop_reason: str | None = None
|
|
226
|
+
# The artifact above is an id the sandbox cannot read back.
|
|
227
|
+
messages: tuple[Any, ...] = ()
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
class CallAborted(RuntimeError):
|
|
231
|
+
"""The call step started but did not finish. `partial`, when known, carries whatever timing
|
|
232
|
+
the call runner already measured — the receipt's `call` field must not be null once the call
|
|
233
|
+
has genuinely started (outbound-channels.md Channel 2, "errored receipt body")."""
|
|
234
|
+
|
|
235
|
+
def __init__(
|
|
236
|
+
self,
|
|
237
|
+
message: str,
|
|
238
|
+
*,
|
|
239
|
+
partial: CallOutcome | None = None,
|
|
240
|
+
code: str = "call_failed",
|
|
241
|
+
) -> None:
|
|
242
|
+
super().__init__(message)
|
|
243
|
+
self.partial = partial
|
|
244
|
+
self.code = code
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
class CallRunner(Protocol):
|
|
248
|
+
async def run(
|
|
249
|
+
self,
|
|
250
|
+
scenario: Scenario,
|
|
251
|
+
runtime: EnvironmentRuntime,
|
|
252
|
+
*,
|
|
253
|
+
world: World | None = None,
|
|
254
|
+
) -> CallOutcome: ...
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
async def _run_call(
|
|
258
|
+
runner: CallRunner,
|
|
259
|
+
scenario: Scenario,
|
|
260
|
+
runtime: EnvironmentRuntime,
|
|
261
|
+
world: World,
|
|
262
|
+
) -> CallOutcome:
|
|
263
|
+
"""Pass the world to text runners while preserving older two-argument integrations.
|
|
264
|
+
|
|
265
|
+
Voice runners do not execute response-carried tools themselves. Hosted HTTP chat runners do,
|
|
266
|
+
and must execute them against the exact leased world that setup/checks observe. The signature
|
|
267
|
+
probe keeps the settled injected-runner seam source compatible for downstream callers while
|
|
268
|
+
allowing that missing context to cross the boundary.
|
|
269
|
+
"""
|
|
270
|
+
run = runner.run
|
|
271
|
+
parameters = inspect.signature(run).parameters.values()
|
|
272
|
+
accepts_world = any(
|
|
273
|
+
parameter.name == "world" or parameter.kind is inspect.Parameter.VAR_KEYWORD
|
|
274
|
+
for parameter in parameters
|
|
275
|
+
)
|
|
276
|
+
if accepts_world:
|
|
277
|
+
return await run(scenario, runtime, world=world)
|
|
278
|
+
return await run(scenario, runtime)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
# --- receipts (outbound-channels.md Channel 2; envelope fields — job_id/attempt_id/digest/etc —
|
|
282
|
+
# are the emitter's concern, not reproduced here) -----------------------------------------------
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
@dataclass(frozen=True)
|
|
286
|
+
class SubGoalResult:
|
|
287
|
+
name: str
|
|
288
|
+
held: bool | None
|
|
289
|
+
reason: str | None
|
|
290
|
+
judged: bool
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
@dataclass(frozen=True)
|
|
294
|
+
class Evaluation:
|
|
295
|
+
name: str
|
|
296
|
+
kind: str # "metric" | "checkpoint"
|
|
297
|
+
reason: str
|
|
298
|
+
score: float | None = None
|
|
299
|
+
passed: bool | None = None
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
@dataclass(frozen=True)
|
|
303
|
+
class CallSummary:
|
|
304
|
+
started_at: str | None
|
|
305
|
+
ended_at: str | None
|
|
306
|
+
duration_ms: int
|
|
307
|
+
turns: int
|
|
308
|
+
transcript_artifact: str | None = None
|
|
309
|
+
recording_artifacts: tuple[str, ...] = ()
|
|
310
|
+
stop_reason: str | None = None
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
@dataclass(frozen=True)
|
|
314
|
+
class ReceiptFailure:
|
|
315
|
+
domain: str
|
|
316
|
+
stage: str
|
|
317
|
+
code: str
|
|
318
|
+
message: str
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
@dataclass(frozen=True)
|
|
322
|
+
class ResultReceipt:
|
|
323
|
+
scenario_key: str
|
|
324
|
+
scenario_id: str
|
|
325
|
+
scenario_attempt: int
|
|
326
|
+
world_index: int | None
|
|
327
|
+
status: str # "passed" | "failed" | "errored" | "skipped"
|
|
328
|
+
sub_goals: tuple[SubGoalResult, ...]
|
|
329
|
+
evaluations: tuple[Evaluation, ...]
|
|
330
|
+
call: CallSummary | None
|
|
331
|
+
failure: ReceiptFailure | None
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
# --- outbound (this module's own minimal sink; see the module docstring's decoupling note) ----
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
class OutboundPort(Protocol):
|
|
338
|
+
async def scenario_started(
|
|
339
|
+
self, *, scenario_key: str, world_index: int, scenario_attempt: int
|
|
340
|
+
) -> None: ...
|
|
341
|
+
|
|
342
|
+
async def scenario_retried(
|
|
343
|
+
self, *, scenario_key: str, from_world: int, to_world: int
|
|
344
|
+
) -> None: ...
|
|
345
|
+
|
|
346
|
+
async def world_unhealthy(self, *, world_index: int, cause: str) -> None: ...
|
|
347
|
+
|
|
348
|
+
async def log(self, *, level: str, message: str) -> None: ...
|
|
349
|
+
|
|
350
|
+
async def receipt(self, receipt: ResultReceipt) -> None: ...
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
# --- failure-code -> FailureDomain, and which codes retry once on a fresh world ----------------
|
|
354
|
+
#
|
|
355
|
+
# `world_unavailable` is domain `environment` per world-handle-interface.md's own errored-receipt
|
|
356
|
+
# table (it overrides the table's default "domain: simulator"). `evidence_missing` is domain
|
|
357
|
+
# simulator but is explicitly carved out as retryable in that same document ("gets the same single
|
|
358
|
+
# retry-on-another-world as a world failure"). `call_failed` (v3.3) and `driver_crashed` (v3.4) are
|
|
359
|
+
# both rows in that same closed table now: `call_failed` is domain infrastructure, retried once
|
|
360
|
+
# like a world failure; `driver_crashed` is domain simulator, not retried — the scheduler's own
|
|
361
|
+
# machinery failing while driving a scenario, distinct from any agent/check/call outcome.
|
|
362
|
+
# `world_pool_exhausted` is NOT a per-scenario receipt code — it is `HostedScheduler.run()`'s own
|
|
363
|
+
# job-abort signal for spine v1.12 §5.4's closed job-level failure vocabulary for stage `running`
|
|
364
|
+
# (domain infrastructure).
|
|
365
|
+
|
|
366
|
+
_CODE_DOMAIN: dict[str, FailureDomain] = {
|
|
367
|
+
"setup_crashed": FailureDomain.SIMULATOR,
|
|
368
|
+
"setup_timeout": FailureDomain.SIMULATOR,
|
|
369
|
+
"ready_timeout": FailureDomain.SIMULATOR,
|
|
370
|
+
"check_timeout": FailureDomain.SIMULATOR,
|
|
371
|
+
"ready_not_ready": FailureDomain.SIMULATOR,
|
|
372
|
+
"ready_broken": FailureDomain.SIMULATOR,
|
|
373
|
+
"check_broken": FailureDomain.SIMULATOR,
|
|
374
|
+
"judge_undecided": FailureDomain.SIMULATOR,
|
|
375
|
+
"evidence_missing": FailureDomain.SIMULATOR,
|
|
376
|
+
"world_usage": FailureDomain.SIMULATOR,
|
|
377
|
+
"world_unavailable": FailureDomain.ENVIRONMENT,
|
|
378
|
+
"state_too_large": FailureDomain.SIMULATOR,
|
|
379
|
+
"call_failed": FailureDomain.INFRASTRUCTURE,
|
|
380
|
+
"target_agent_stalled": FailureDomain.AGENT,
|
|
381
|
+
"target_agent_tool_failed": FailureDomain.AGENT,
|
|
382
|
+
"simulator_stalled": FailureDomain.SIMULATOR,
|
|
383
|
+
"driver_crashed": FailureDomain.SIMULATOR,
|
|
384
|
+
"world_pool_exhausted": FailureDomain.INFRASTRUCTURE,
|
|
385
|
+
}
|
|
386
|
+
_RETRYABLE_CODES = frozenset(
|
|
387
|
+
{"evidence_missing", "target_agent_stalled", "simulator_stalled"}
|
|
388
|
+
)
|
|
389
|
+
|
|
390
|
+
# hosted-execution-seams.md v1.13 §5.4/§2f: the closed provisioner build/run failure-code table --
|
|
391
|
+
# these used to be discarded at the reset()/provision() seam (caught as a bare `Exception`, only
|
|
392
|
+
# `str()` surviving into `world_unhealthy.cause`), so a deterministic `environment`/`agent` fault
|
|
393
|
+
# (never retried) was re-reported as `world_pool_exhausted`/infrastructure and burned every
|
|
394
|
+
# whole-job retry on a failure that repeats identically. v1.15: the PRODUCER now resolves
|
|
395
|
+
# `spawn_failed`'s managed-vs-source split at the raise site (`ProcessRuntimeError.domain`) --
|
|
396
|
+
# this module reads that carried domain first; `SECTION_2F_DOMAIN` (imported from
|
|
397
|
+
# `process_runtime.py`, the codes' own home) is consulted only for the rare error that reaches
|
|
398
|
+
# here with no carried domain, and that fallback is logged so a silent re-guess never hides again.
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _resolve_2f_domain(
|
|
402
|
+
code: str | None,
|
|
403
|
+
domain: FailureDomain | None,
|
|
404
|
+
) -> tuple[str, FailureDomain] | None:
|
|
405
|
+
"""v1.15 §2f: pair a code with its resolved domain. `domain` should be the value CARRIED by a
|
|
406
|
+
typed provisioner error (`ProcessRuntimeError.domain`); `None` here falls back to the closed
|
|
407
|
+
code->domain map (and logs it) rather than silently re-guessing. Returns `None` outright for
|
|
408
|
+
no code, or a code outside the §2f table (`internal_*` etc.) — callers already treat that the
|
|
409
|
+
same as "not a §2f error."
|
|
410
|
+
"""
|
|
411
|
+
if code is None or code not in SECTION_2F_DOMAIN:
|
|
412
|
+
return None
|
|
413
|
+
if domain is not None:
|
|
414
|
+
return code, domain
|
|
415
|
+
logger.warning(
|
|
416
|
+
"process_runtime error %r crossed the §4 seam with no carried domain; using the §2f "
|
|
417
|
+
"fallback map (%s)",
|
|
418
|
+
code,
|
|
419
|
+
SECTION_2F_DOMAIN[code].value,
|
|
420
|
+
)
|
|
421
|
+
return code, SECTION_2F_DOMAIN[code]
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
# v1.13: only these two domains are "never retried" -- a uniform §2f code across every unhealthy
|
|
425
|
+
# world in one of them surfaces as that code+domain; anything else (mixed codes, or any
|
|
426
|
+
# infrastructure-domain fault) stays `world_pool_exhausted` exactly as before.
|
|
427
|
+
_SECTION_2F_NEVER_RETRIED = frozenset({FailureDomain.ENVIRONMENT, FailureDomain.AGENT})
|
|
428
|
+
|
|
429
|
+
# The one `OutboundPort` failure class that is NOT best-effort -- outbound-channels.md:
|
|
430
|
+
# 401/403 -> "stop emitting, exit code 3 ... never an infra retry," and the same latch also covers
|
|
431
|
+
# a 404-exhausted channel and a 409 attempt-supersession (outbound.py's `ChannelState`: "a fence in
|
|
432
|
+
# substance"). Letting `_emit`/`_log`/`mark_unhealthy` swallow any of these the same way they
|
|
433
|
+
# swallow a transport hiccup ran a superseded attempt's entire scenario set after the platform had
|
|
434
|
+
# already fenced or superseded it.
|
|
435
|
+
_FATAL_OUTBOUND: tuple[type[Exception], ...] = (
|
|
436
|
+
HostedFencedError,
|
|
437
|
+
HostedChannelFailedError,
|
|
438
|
+
HostedAttemptSupersededError,
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
# M13: an exception/overrun outcome leaves the world half-applied — world-handle-interface.md's
|
|
442
|
+
# return-conventions rule is "the world is discarded and re-provisioned (a half-applied world is
|
|
443
|
+
# never reused)." `ready_not_ready` is deliberately excluded: a precondition failing on the shared
|
|
444
|
+
# sealed baseline is a clean verdict, not an exception, so the world itself is still fine.
|
|
445
|
+
_DISCARD_ON_ERROR_CODES = frozenset(
|
|
446
|
+
{
|
|
447
|
+
"setup_crashed",
|
|
448
|
+
"setup_timeout",
|
|
449
|
+
"ready_timeout",
|
|
450
|
+
"check_timeout",
|
|
451
|
+
"ready_broken",
|
|
452
|
+
"check_broken",
|
|
453
|
+
"world_usage",
|
|
454
|
+
"state_too_large",
|
|
455
|
+
}
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
SETUP_TIMEOUT_SECONDS = 60.0
|
|
459
|
+
READY_TIMEOUT_SECONDS = 15.0
|
|
460
|
+
CHECK_TIMEOUT_SECONDS = 60.0
|
|
461
|
+
|
|
462
|
+
_MESSAGE_LIMIT = 2000 # matches the Call.result/error truncation convention (world-handle-interface.md).
|
|
463
|
+
_CAUSE_LIMIT = (
|
|
464
|
+
200 # outbound-channels.md Channel 1: `world_unhealthy.cause` free text <=200.
|
|
465
|
+
)
|
|
466
|
+
_USERINFO_PATTERN = re.compile(r"://[^@/]+@")
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def _is_retryable(code: str) -> bool:
|
|
470
|
+
return _CODE_DOMAIN[code] in (
|
|
471
|
+
FailureDomain.ENVIRONMENT,
|
|
472
|
+
FailureDomain.INFRASTRUCTURE,
|
|
473
|
+
) or (code in _RETRYABLE_CODES)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _truncate(text: str, limit: int = _MESSAGE_LIMIT) -> str:
|
|
477
|
+
if len(text) <= limit:
|
|
478
|
+
return text
|
|
479
|
+
return text[: limit - len("…[truncated]")] + "…[truncated]"
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def _sanitize_cause(message: str) -> str:
|
|
483
|
+
# M7: `cause` is capped at 200 chars and must never carry endpoint credentials — postgres
|
|
484
|
+
# error strings routinely embed the DSN (`postgresql://user:pw@host/db`).
|
|
485
|
+
return _truncate(_USERINFO_PATTERN.sub("://***@", message), _CAUSE_LIMIT)
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _failure(code: str, message: str) -> ReceiptFailure:
|
|
489
|
+
return ReceiptFailure(
|
|
490
|
+
domain=_CODE_DOMAIN[code].value,
|
|
491
|
+
stage=HarnessStage.RUNNING.value,
|
|
492
|
+
code=code,
|
|
493
|
+
message=_truncate(message),
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
# --- return-convention classification (world-handle-interface.md "Return conventions") --------
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
@dataclass(frozen=True)
|
|
501
|
+
class _Verdict:
|
|
502
|
+
held: bool
|
|
503
|
+
reason: str | None
|
|
504
|
+
broken: bool
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _classify_ready(value: object) -> _Verdict:
|
|
508
|
+
if value is None or value is True:
|
|
509
|
+
return _Verdict(True, None, False)
|
|
510
|
+
if isinstance(value, str):
|
|
511
|
+
if value.strip() == "":
|
|
512
|
+
return _Verdict(True, None, False)
|
|
513
|
+
return _Verdict(False, value, False)
|
|
514
|
+
# Bare False or any other value -> broken. checks.py's `run_world_check` treats a non-None,
|
|
515
|
+
# non-string ready() answer the same way; a scenario hitting this cannot be told apart from a
|
|
516
|
+
# buggy ready.py, which is why it is `ready_broken` rather than a clean not-ready verdict.
|
|
517
|
+
return _Verdict(False, None, True)
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def _classify_check(value: object) -> _Verdict:
|
|
521
|
+
if value is None or value is True:
|
|
522
|
+
return _Verdict(True, None, False)
|
|
523
|
+
if isinstance(value, str):
|
|
524
|
+
if value.strip() == "":
|
|
525
|
+
return _Verdict(True, None, False)
|
|
526
|
+
return _Verdict(False, value, False)
|
|
527
|
+
if value is False:
|
|
528
|
+
# An agent result ("the agent did something wrong"), not a broken check — matches
|
|
529
|
+
# checks.py's `Outcome(name, False, "False")`.
|
|
530
|
+
return _Verdict(False, "False", False)
|
|
531
|
+
return _Verdict(False, None, True)
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def _sub_goal_reason(goal: SubGoal, verdict: _Verdict) -> str | None:
|
|
535
|
+
"""What to show a reader for this sub-goal, on a pass as much as on a failure.
|
|
536
|
+
|
|
537
|
+
A check returns nothing when it holds, which left every passing sub-goal with an empty hover
|
|
538
|
+
and no way to tell a real pass from one nobody wrote a check for. The authored description of
|
|
539
|
+
what the sub-goal means is the honest thing to show there: it says what was verified without
|
|
540
|
+
claiming evidence the check never returned. A bare ``False`` is the other end of the same
|
|
541
|
+
problem -- the reason read literally "False" -- so it gets the description too.
|
|
542
|
+
"""
|
|
543
|
+
what = str(getattr(goal, "what", "") or "").strip().rstrip(".")
|
|
544
|
+
if verdict.held:
|
|
545
|
+
return f"Held: {what}." if what else "Held. The check found nothing wrong."
|
|
546
|
+
if verdict.reason and verdict.reason.strip() and verdict.reason != "False":
|
|
547
|
+
return verdict.reason
|
|
548
|
+
return f"Did not hold: {what}." if what else None
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
# --- phase execution: budget + exception classification ---------------------------------------
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
class _PhaseTimeout(Exception):
|
|
555
|
+
def __init__(self, phase: str) -> None:
|
|
556
|
+
super().__init__(phase)
|
|
557
|
+
self.phase = phase
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
class _PhaseNeverStarted(Exception):
|
|
561
|
+
"""R1: the phase's own worker thread had not even started running when its budget elapsed —
|
|
562
|
+
the dedicated executor was saturated, not the phase itself overrunning. Must not read as a
|
|
563
|
+
genuine timeout (which discards the world); the world did nothing wrong here."""
|
|
564
|
+
|
|
565
|
+
def __init__(self, phase: str) -> None:
|
|
566
|
+
super().__init__(phase)
|
|
567
|
+
self.phase = phase
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
class _PhaseWorldGone(Exception):
|
|
571
|
+
def __init__(self, phase: str, cause: BaseException) -> None:
|
|
572
|
+
super().__init__(f"{phase}: {cause}")
|
|
573
|
+
self.cause = cause
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
class _PhaseMisuse(Exception):
|
|
577
|
+
def __init__(self, phase: str, cause: BaseException) -> None:
|
|
578
|
+
super().__init__(f"{phase}: {cause}")
|
|
579
|
+
self.cause = cause
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
class _PhaseStateTooLarge(Exception):
|
|
583
|
+
def __init__(self, phase: str, cause: BaseException) -> None:
|
|
584
|
+
super().__init__(f"{phase}: {cause}")
|
|
585
|
+
self.cause = cause
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
class _PhaseCrashed(Exception):
|
|
589
|
+
def __init__(self, phase: str, cause: BaseException) -> None:
|
|
590
|
+
super().__init__(f"{phase}: {cause}")
|
|
591
|
+
self.cause = cause
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
async def _invoke(
|
|
595
|
+
fn: Callable[..., object],
|
|
596
|
+
*args: object,
|
|
597
|
+
timeout: float,
|
|
598
|
+
phase: str,
|
|
599
|
+
executor: ThreadPoolExecutor,
|
|
600
|
+
) -> object:
|
|
601
|
+
# R1: a `threading.Event` set as the thread body's first statement — the only way to tell
|
|
602
|
+
# "the phase ran past its budget" (genuine overrun, world half-applied) apart from "the
|
|
603
|
+
# phase's thread was still queued behind others when the budget elapsed" (the scheduler's own
|
|
604
|
+
# executor was saturated; the world itself never touched anything).
|
|
605
|
+
started_flag = threading.Event()
|
|
606
|
+
|
|
607
|
+
def _run() -> object:
|
|
608
|
+
started_flag.set()
|
|
609
|
+
return fn(*args)
|
|
610
|
+
|
|
611
|
+
async def _call() -> object:
|
|
612
|
+
# B4: real scenario code (`setup`/`ready`/`check`) is synchronous, blocking psycopg calls
|
|
613
|
+
# — it must never run directly on the event loop, or the timeout below is purely
|
|
614
|
+
# decorative and every other world stalls with it. Dispatched to the scheduler's own
|
|
615
|
+
# dedicated executor (world-handle-interface.md: "one worker thread per world"; R1 —
|
|
616
|
+
# never the loop's default executor, which the provider's own `to_thread` calls also
|
|
617
|
+
# use). If the thread's own return value is itself awaitable (scenario code that is
|
|
618
|
+
# `async def`, reached indirectly through a sync wrapper), that coroutine is driven on
|
|
619
|
+
# the event loop afterward, where real suspension/cancellation actually works — this is
|
|
620
|
+
# the kept "awaitable" branch.
|
|
621
|
+
loop = asyncio.get_running_loop()
|
|
622
|
+
result = await loop.run_in_executor(executor, _run)
|
|
623
|
+
if inspect.isawaitable(result):
|
|
624
|
+
result = await result
|
|
625
|
+
return result
|
|
626
|
+
|
|
627
|
+
loop = asyncio.get_running_loop()
|
|
628
|
+
started = loop.time()
|
|
629
|
+
try:
|
|
630
|
+
return await asyncio.wait_for(_call(), timeout=timeout)
|
|
631
|
+
except asyncio.TimeoutError as exc:
|
|
632
|
+
# m4: `asyncio.TimeoutError is TimeoutError` on 3.11 — a psycopg statement timeout raised
|
|
633
|
+
# INSIDE `fn` looks identical to `wait_for`'s own deadline unless the elapsed time is
|
|
634
|
+
# actually checked. If the budget did not genuinely elapse, this was `fn`'s own timeout
|
|
635
|
+
# bubbling through — a broken phase, not a budget overrun.
|
|
636
|
+
if loop.time() - started < timeout:
|
|
637
|
+
raise _PhaseCrashed(phase, exc) from exc
|
|
638
|
+
if not started_flag.is_set():
|
|
639
|
+
raise _PhaseNeverStarted(phase) from exc
|
|
640
|
+
# B4: `wait_for`'s cancellation stops US from waiting on the thread, not the thread
|
|
641
|
+
# itself — psycopg in-flight cancellation is not wired here (P11 follow-up; recorded in
|
|
642
|
+
# the fixer report). The thread is abandoned, bounded by scenario count per the contract's
|
|
643
|
+
# own accepted tradeoff; its world is discarded rather than reused (M13).
|
|
644
|
+
raise _PhaseTimeout(phase) from exc
|
|
645
|
+
except WorldUnavailable as exc:
|
|
646
|
+
raise _PhaseWorldGone(phase, exc) from exc
|
|
647
|
+
except WorldStateTooLarge as exc:
|
|
648
|
+
raise _PhaseStateTooLarge(phase, exc) from exc
|
|
649
|
+
except (
|
|
650
|
+
WorldReadOnly,
|
|
651
|
+
WorldReservedName,
|
|
652
|
+
WorldQueryRejected,
|
|
653
|
+
WorldUsageError,
|
|
654
|
+
) as exc:
|
|
655
|
+
raise _PhaseMisuse(phase, exc) from exc
|
|
656
|
+
except WorldError as exc:
|
|
657
|
+
# m5: catches any WorldError subclass not special-cased above (world/errors.py's own
|
|
658
|
+
# base, kept exactly for "route 'scenario code misused the handle' to one outcome without
|
|
659
|
+
# naming all six") — a future seventh subclass lands here instead of silently falling into
|
|
660
|
+
# the generic crash classification below.
|
|
661
|
+
raise _PhaseMisuse(phase, exc) from exc
|
|
662
|
+
except Exception as exc:
|
|
663
|
+
raise _PhaseCrashed(phase, exc) from exc
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
_CRASH_CODE_BY_PHASE = {
|
|
667
|
+
"setup": "setup_crashed",
|
|
668
|
+
"ready": "ready_broken",
|
|
669
|
+
"check": "check_broken",
|
|
670
|
+
}
|
|
671
|
+
_TIMEOUT_CODE_BY_PHASE = {
|
|
672
|
+
"setup": "setup_timeout",
|
|
673
|
+
"ready": "ready_timeout",
|
|
674
|
+
"check": "check_timeout",
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
@dataclass(frozen=True)
|
|
679
|
+
class _PhaseResult:
|
|
680
|
+
value: object
|
|
681
|
+
failure: ReceiptFailure | None
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
async def _run_phase(
|
|
685
|
+
fn: Callable[..., object],
|
|
686
|
+
*args: object,
|
|
687
|
+
timeout: float,
|
|
688
|
+
phase: str,
|
|
689
|
+
executor: ThreadPoolExecutor,
|
|
690
|
+
) -> _PhaseResult:
|
|
691
|
+
try:
|
|
692
|
+
value = await _invoke(
|
|
693
|
+
fn, *args, timeout=timeout, phase=phase, executor=executor
|
|
694
|
+
)
|
|
695
|
+
return _PhaseResult(value, None)
|
|
696
|
+
except _PhaseNeverStarted:
|
|
697
|
+
# R1: not the phase's fault and not the world's — the scheduler's own thread pool
|
|
698
|
+
# couldn't service it in time. `driver_crashed` is not in `_DISCARD_ON_ERROR_CODES`, so
|
|
699
|
+
# this releases the world rather than discarding a perfectly healthy one.
|
|
700
|
+
return _PhaseResult(
|
|
701
|
+
None,
|
|
702
|
+
_failure(
|
|
703
|
+
"driver_crashed",
|
|
704
|
+
f"{phase} never started before its budget elapsed (thread pool saturated)",
|
|
705
|
+
),
|
|
706
|
+
)
|
|
707
|
+
except _PhaseTimeout:
|
|
708
|
+
return _PhaseResult(
|
|
709
|
+
None,
|
|
710
|
+
_failure(_TIMEOUT_CODE_BY_PHASE[phase], f"{phase} exceeded its budget"),
|
|
711
|
+
)
|
|
712
|
+
except _PhaseWorldGone as exc:
|
|
713
|
+
return _PhaseResult(None, _failure("world_unavailable", str(exc.cause)))
|
|
714
|
+
except _PhaseMisuse as exc:
|
|
715
|
+
return _PhaseResult(None, _failure("world_usage", str(exc.cause)))
|
|
716
|
+
except _PhaseStateTooLarge as exc:
|
|
717
|
+
return _PhaseResult(None, _failure("state_too_large", str(exc.cause)))
|
|
718
|
+
except _PhaseCrashed as exc:
|
|
719
|
+
return _PhaseResult(
|
|
720
|
+
None,
|
|
721
|
+
_failure(
|
|
722
|
+
_CRASH_CODE_BY_PHASE[phase], f"{type(exc.cause).__name__}: {exc.cause}"
|
|
723
|
+
),
|
|
724
|
+
)
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
# --- the world pool -----------------------------------------------------------------------------
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
class NoWorldsAvailable(RuntimeError):
|
|
731
|
+
"""Every provisioned world is down and none is currently recoverable (spine v1.12 §5.4:
|
|
732
|
+
"if ready worlds reach 0 the job FAILS in stage running, domain infrastructure" — declared
|
|
733
|
+
only after in-flight re-provisioning completes without restoring a world), OR the pool has
|
|
734
|
+
been closed (R5: `reason="closed"`).
|
|
735
|
+
|
|
736
|
+
v1.13 §5.4: `code`/`domain` carry a uniform §2f never-retried code when every unhealthy
|
|
737
|
+
world's last re-provision attempt agreed on one — `None` (the default) means the caller falls
|
|
738
|
+
back to the generic `world_pool_exhausted`/infrastructure abort, exactly as before."""
|
|
739
|
+
|
|
740
|
+
def __init__(
|
|
741
|
+
self,
|
|
742
|
+
message: str,
|
|
743
|
+
*,
|
|
744
|
+
reason: str = "exhausted",
|
|
745
|
+
code: str | None = None,
|
|
746
|
+
domain: FailureDomain | None = None,
|
|
747
|
+
) -> None:
|
|
748
|
+
super().__init__(message)
|
|
749
|
+
self.reason = reason
|
|
750
|
+
self.code = code
|
|
751
|
+
self.domain = domain
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
_RECONCILE_MAX_ATTEMPTS = 3
|
|
755
|
+
_RECONCILE_BACKOFF_SECONDS = (0.05, 0.1)
|
|
756
|
+
_LEASE_POLL_INTERVAL_SECONDS = 0.02
|
|
757
|
+
|
|
758
|
+
|
|
759
|
+
class WorldPool:
|
|
760
|
+
"""Leases/releases the W worlds `provisioner.provision()` returns, resets one to pristine on
|
|
761
|
+
every lease (spine §4.2), and reconciles an unhealthy world back in via `provision()` again
|
|
762
|
+
(§4 rule 1: "a sick world mid-job is recovered by calling `provision` again") — in the
|
|
763
|
+
background, so a lease elsewhere never blocks on someone else's recovery.
|
|
764
|
+
|
|
765
|
+
B1/B2/M6 (spine v1.12 §4.5b): the provider port is NOT reentrant — at most one
|
|
766
|
+
`provision`/`reset`/`healthy`/`close` call is ever in flight, serialized by `_provider_lock`
|
|
767
|
+
(R13: `healthy` writes — it demotes state — so v1.12 folded it into the same serialized set
|
|
768
|
+
that provision/reset/close were already in; it is no longer treated as a read-only probe
|
|
769
|
+
exempt from the lock). A demotion that lands while a reconcile is already running is coalesced
|
|
770
|
+
into a trailing pass rather than a second concurrent `provision()` call.
|
|
771
|
+
"""
|
|
772
|
+
|
|
773
|
+
def __init__(
|
|
774
|
+
self,
|
|
775
|
+
provisioner: WorldProvisioner,
|
|
776
|
+
*,
|
|
777
|
+
bundle: Any,
|
|
778
|
+
source: Path,
|
|
779
|
+
bundle_dir: Path,
|
|
780
|
+
work_directory: Path,
|
|
781
|
+
instances: int,
|
|
782
|
+
outbound: OutboundPort | None = None,
|
|
783
|
+
) -> None:
|
|
784
|
+
self._provisioner = provisioner
|
|
785
|
+
self._bundle = bundle
|
|
786
|
+
self._source = source
|
|
787
|
+
self._bundle_dir = bundle_dir
|
|
788
|
+
self._work_directory = work_directory
|
|
789
|
+
self._instances = instances
|
|
790
|
+
self._outbound = outbound
|
|
791
|
+
|
|
792
|
+
self._runtimes: dict[int, EnvironmentRuntime] = {}
|
|
793
|
+
self._available: set[int] = set()
|
|
794
|
+
self._leased: set[int] = set()
|
|
795
|
+
self._down: set[int] = set()
|
|
796
|
+
self._fresh: set[int] = (
|
|
797
|
+
set()
|
|
798
|
+
) # m9: provisioned/recovered but never yet leased/reset
|
|
799
|
+
self._effective_size = 0 # R2: the achieved world count `start()` settled on
|
|
800
|
+
# The §2f (code, domain) pair (or `None`) behind the most recent demotion/reconcile-failure
|
|
801
|
+
# for a down world index -- read by `lease()`'s exhaustion check to decide whether a
|
|
802
|
+
# uniform never-retried code can surface instead of the generic `world_pool_exhausted`.
|
|
803
|
+
# `domain` is the CARRIED value off the typed error (v1.15), captured once here rather than
|
|
804
|
+
# re-derived later from the code alone.
|
|
805
|
+
self._down_codes: dict[int, tuple[str, FailureDomain] | None] = {}
|
|
806
|
+
self._fenced: BaseException | None = (
|
|
807
|
+
None # latched by mark_fenced(), never cleared
|
|
808
|
+
)
|
|
809
|
+
|
|
810
|
+
# m1: `asyncio.Condition` (not a manual `Event` + `clear()`) — waiting and notifying share
|
|
811
|
+
# one lock, so there is no window between releasing a lock and clearing a flag for a
|
|
812
|
+
# `set()` to land in and be silently lost.
|
|
813
|
+
self._state_lock = asyncio.Condition()
|
|
814
|
+
self._provider_lock = asyncio.Lock()
|
|
815
|
+
self._reconcile_task: asyncio.Task[None] | None = None
|
|
816
|
+
self._reconcile_pending = False
|
|
817
|
+
self._started = False
|
|
818
|
+
self._closing = (
|
|
819
|
+
False # R4: set at the top of close() -- lets an in-flight reconcile bail
|
|
820
|
+
)
|
|
821
|
+
# between attempts instead of burning close()'s wait budget on a pool being torn down.
|
|
822
|
+
self._closed = (
|
|
823
|
+
False # Set once close() STARTS -- latches provision()/lease() out for
|
|
824
|
+
)
|
|
825
|
+
# good immediately, independent of whether teardown itself has finished.
|
|
826
|
+
self._teardown_task: asyncio.Task[None] | None = None # the shared, retry-safe
|
|
827
|
+
# teardown -- see close()'s own comment for why idempotency lives here now, not on
|
|
828
|
+
# `_closed`.
|
|
829
|
+
|
|
830
|
+
@property
|
|
831
|
+
def effective_size(self) -> int:
|
|
832
|
+
"""R2: the world count `start()` actually achieved — may be less than the requested
|
|
833
|
+
`instances` on a legitimate degrade (conformance-gate failure, `fixed_port`). P10 sizes
|
|
834
|
+
`parallelism_degraded` and anything else that needs "how many worlds do we really have"
|
|
835
|
+
off this, never off the originally requested `instances`."""
|
|
836
|
+
return self._effective_size
|
|
837
|
+
|
|
838
|
+
@property
|
|
839
|
+
def fenced(self) -> BaseException | None:
|
|
840
|
+
"""The first fatal `OutboundPort` exception (401/403 -> `HostedFencedError`, a
|
|
841
|
+
404-exhausted channel -> `HostedChannelFailedError`, or a 409 attempt-supersession ->
|
|
842
|
+
`HostedAttemptSupersededError`) observed anywhere along this pool's own emit paths.
|
|
843
|
+
`HostedScheduler` polls this at the same points it polls `cancel_requested` to stop
|
|
844
|
+
leasing/launching further scenarios once set."""
|
|
845
|
+
return self._fenced
|
|
846
|
+
|
|
847
|
+
def mark_fenced(self, exc: BaseException) -> None:
|
|
848
|
+
if self._fenced is None:
|
|
849
|
+
self._fenced = exc
|
|
850
|
+
|
|
851
|
+
@property
|
|
852
|
+
def size(self) -> int:
|
|
853
|
+
return len(self._runtimes)
|
|
854
|
+
|
|
855
|
+
async def start(self) -> list[EnvironmentRuntime]:
|
|
856
|
+
if self._started:
|
|
857
|
+
# m10: a second call would re-provision behind every already-leased world's back.
|
|
858
|
+
raise RuntimeError("WorldPool.start() called more than once")
|
|
859
|
+
self._started = True
|
|
860
|
+
|
|
861
|
+
async with self._provider_lock:
|
|
862
|
+
runtimes = await self._provisioner.provision(
|
|
863
|
+
self._bundle,
|
|
864
|
+
source=self._source,
|
|
865
|
+
bundle_dir=self._bundle_dir,
|
|
866
|
+
work_directory=self._work_directory,
|
|
867
|
+
instances=self._instances,
|
|
868
|
+
)
|
|
869
|
+
|
|
870
|
+
# R2 (spine v1.12 §4's conformance gate / `fixed_port`): `provision()` legitimately
|
|
871
|
+
# returns FEWER than `instances` worlds — "Fail → effective parallelism 1 +
|
|
872
|
+
# parallelism_degraded ... Loud, never silent," not a failure this pool should raise on.
|
|
873
|
+
# Reject only a genuinely malformed result: zero worlds, duplicates, a non-contiguous
|
|
874
|
+
# index set, or more worlds than were ever requested.
|
|
875
|
+
indices = {runtime.world_index for runtime in runtimes}
|
|
876
|
+
if (
|
|
877
|
+
not runtimes
|
|
878
|
+
or len(runtimes) != len(indices)
|
|
879
|
+
or indices != set(range(len(runtimes)))
|
|
880
|
+
or len(runtimes) > self._instances
|
|
881
|
+
):
|
|
882
|
+
# m10/R2: spine §4 — "ordered by world_index" and contiguous from 0 (what
|
|
883
|
+
# `range(effective_instances)` on the provider side guarantees).
|
|
884
|
+
raise RuntimeError(
|
|
885
|
+
f"provision() returned world_index set {sorted(indices)}, expected a contiguous "
|
|
886
|
+
f"0..N-1 subset of 0..{self._instances - 1}"
|
|
887
|
+
)
|
|
888
|
+
self._effective_size = len(runtimes)
|
|
889
|
+
|
|
890
|
+
async with self._state_lock:
|
|
891
|
+
for runtime in runtimes:
|
|
892
|
+
self._runtimes[runtime.world_index] = runtime
|
|
893
|
+
if runtime.state in (RuntimeState.READY, RuntimeState.PREPARING):
|
|
894
|
+
# m10: never hand out a world provision() itself returned UNHEALTHY. A
|
|
895
|
+
# PREPARING world legitimately demotes straight to UNHEALTHY on a failed first
|
|
896
|
+
# reset/probe (spine v1.12 §3's preparing->unhealthy transition) -- lease()'s
|
|
897
|
+
# own health gate covers that case; nothing extra is needed here.
|
|
898
|
+
self._available.add(runtime.world_index)
|
|
899
|
+
if runtime.state is RuntimeState.READY:
|
|
900
|
+
self._fresh.add(runtime.world_index)
|
|
901
|
+
else:
|
|
902
|
+
self._down.add(runtime.world_index)
|
|
903
|
+
self._state_lock.notify_all()
|
|
904
|
+
return runtimes
|
|
905
|
+
|
|
906
|
+
def _reconcile_in_flight(self) -> bool:
|
|
907
|
+
return self._reconcile_task is not None and not self._reconcile_task.done()
|
|
908
|
+
|
|
909
|
+
async def _wait_bounded(self, *, poll: bool) -> None:
|
|
910
|
+
if not poll:
|
|
911
|
+
await self._state_lock.wait()
|
|
912
|
+
return
|
|
913
|
+
try:
|
|
914
|
+
await asyncio.wait_for(
|
|
915
|
+
self._state_lock.wait(), timeout=_LEASE_POLL_INTERVAL_SECONDS
|
|
916
|
+
)
|
|
917
|
+
except asyncio.TimeoutError:
|
|
918
|
+
pass # `Condition.wait()` reacquires the lock before propagating even on timeout.
|
|
919
|
+
|
|
920
|
+
async def lease(
|
|
921
|
+
self,
|
|
922
|
+
*,
|
|
923
|
+
exclude: frozenset[int] = frozenset(),
|
|
924
|
+
abandon: Callable[[], bool] | None = None,
|
|
925
|
+
) -> tuple[int, EnvironmentRuntime] | None:
|
|
926
|
+
"""Returns `None` if `abandon()` reports true while this call was queued (B5) — the caller
|
|
927
|
+
never received a world, so there is nothing to release."""
|
|
928
|
+
while True:
|
|
929
|
+
# R5: latched once close() has run — a lease past that point must never spawn a
|
|
930
|
+
# `reset()`/`healthy()` call against a provider that may already be hard-cleaned.
|
|
931
|
+
if self._closed:
|
|
932
|
+
raise NoWorldsAvailable("world pool is closed", reason="closed")
|
|
933
|
+
if abandon is not None and abandon():
|
|
934
|
+
return None
|
|
935
|
+
|
|
936
|
+
async with self._state_lock:
|
|
937
|
+
candidates = self._available - exclude
|
|
938
|
+
if candidates:
|
|
939
|
+
world_index = min(candidates)
|
|
940
|
+
self._available.discard(world_index)
|
|
941
|
+
self._leased.add(world_index)
|
|
942
|
+
skip_reset = world_index in self._fresh # m9
|
|
943
|
+
self._fresh.discard(world_index)
|
|
944
|
+
else:
|
|
945
|
+
# Not just "every world is down" (the plain retry-exhausted case) — a world
|
|
946
|
+
# excluded for this lease (a same-scenario retry avoiding its failed world)
|
|
947
|
+
# can never satisfy `candidates` again no matter how long we wait, so it must
|
|
948
|
+
# count as unusable here too or a single-world pool's retry blocks forever.
|
|
949
|
+
usable = set(self._runtimes) - exclude
|
|
950
|
+
if not (usable - self._down):
|
|
951
|
+
# M9/R10 (spine v1.12 §5.4): declare exhaustion only once no reconcile is
|
|
952
|
+
# in flight or about to be — never on an instantaneous snapshot of world
|
|
953
|
+
# states. `_reconcile_pending` (set inside `mark_unhealthy`'s own critical
|
|
954
|
+
# section, R10) covers the gap between a demotion and its reconcile task
|
|
955
|
+
# actually existing.
|
|
956
|
+
if self._reconcile_in_flight() or self._reconcile_pending:
|
|
957
|
+
await self._wait_bounded(poll=abandon is not None)
|
|
958
|
+
continue
|
|
959
|
+
# v1.13 §5.4: a uniform §2f never-retried code across every
|
|
960
|
+
# currently-unhealthy world surfaces AS that code+domain; mixed codes, an
|
|
961
|
+
# unrecorded (non-§2f) cause, or any infrastructure-domain code all fall
|
|
962
|
+
# back to the generic `world_pool_exhausted` exactly as before. The
|
|
963
|
+
# uniformity set is `self._down` -- every unhealthy world in the pool, not
|
|
964
|
+
# just `usable` (runtimes minus this call's `exclude`) -- a world excluded
|
|
965
|
+
# because it is the scenario's own just-failed world is still part of "every
|
|
966
|
+
# unhealthy world" the spec means; narrowing to `usable` would let that
|
|
967
|
+
# excluded world's own (possibly untyped) failure escape the check entirely.
|
|
968
|
+
codes = {self._down_codes.get(index) for index in self._down}
|
|
969
|
+
code = domain = None
|
|
970
|
+
if len(codes) == 1:
|
|
971
|
+
(only,) = codes
|
|
972
|
+
if only is not None:
|
|
973
|
+
only_code, only_domain = only
|
|
974
|
+
if only_domain in _SECTION_2F_NEVER_RETRIED:
|
|
975
|
+
code, domain = only_code, only_domain
|
|
976
|
+
raise NoWorldsAvailable(
|
|
977
|
+
f"{len(self._down)}/{len(self._runtimes)} worlds unhealthy, "
|
|
978
|
+
f"none available outside {sorted(exclude)}",
|
|
979
|
+
code=code,
|
|
980
|
+
domain=domain,
|
|
981
|
+
)
|
|
982
|
+
await self._wait_bounded(poll=abandon is not None)
|
|
983
|
+
continue
|
|
984
|
+
|
|
985
|
+
reset_exc: Exception | None = None
|
|
986
|
+
probed_runtime: EnvironmentRuntime | None = None
|
|
987
|
+
if not skip_reset:
|
|
988
|
+
async with self._provider_lock:
|
|
989
|
+
if self._closed:
|
|
990
|
+
# close() can win the `_provider_lock` FIFO queue against a lease already
|
|
991
|
+
# past the top-of-loop `_closed` check -- re-check on the inside too, or
|
|
992
|
+
# this lease drives reset() against a provider close() may already be
|
|
993
|
+
# hard-cleaning.
|
|
994
|
+
raise NoWorldsAvailable("world pool is closed", reason="closed")
|
|
995
|
+
runtime = self._runtimes.get(world_index)
|
|
996
|
+
probed_runtime = runtime
|
|
997
|
+
if runtime is not None:
|
|
998
|
+
try:
|
|
999
|
+
await self._provisioner.reset(
|
|
1000
|
+
runtime, work_directory=self._work_directory
|
|
1001
|
+
)
|
|
1002
|
+
except Exception as exc: # noqa: BLE001 - B3: must never leak out of lease()
|
|
1003
|
+
reset_exc = exc
|
|
1004
|
+
|
|
1005
|
+
is_healthy = False
|
|
1006
|
+
if reset_exc is None:
|
|
1007
|
+
# M2: `healthy()` is called unconditionally after reset — including the m9 fast
|
|
1008
|
+
# path, which skips only the (expensive) reset call, never the readiness check.
|
|
1009
|
+
# R13 (spine v1.12 §4.5b): `healthy` now rides the port's non-reentrancy rule too,
|
|
1010
|
+
# so it goes under `_provider_lock` like reset/provision/close.
|
|
1011
|
+
async with self._provider_lock:
|
|
1012
|
+
if self._closed:
|
|
1013
|
+
raise NoWorldsAvailable(
|
|
1014
|
+
"world pool is closed", reason="closed"
|
|
1015
|
+
) # same re-check as above
|
|
1016
|
+
runtime = self._runtimes.get(world_index)
|
|
1017
|
+
probed_runtime = runtime
|
|
1018
|
+
if runtime is not None:
|
|
1019
|
+
try:
|
|
1020
|
+
is_healthy = await self._provisioner.healthy(
|
|
1021
|
+
runtime, work_directory=self._work_directory
|
|
1022
|
+
)
|
|
1023
|
+
except Exception as exc: # noqa: BLE001
|
|
1024
|
+
reset_exc = exc
|
|
1025
|
+
|
|
1026
|
+
async with self._state_lock:
|
|
1027
|
+
# m2/R14: re-read after the awaited provider calls — a concurrent reconcile may
|
|
1028
|
+
# have replaced or dropped this index's `EnvironmentRuntime` while lease() awaited.
|
|
1029
|
+
# `is_healthy` was computed against `probed_runtime` specifically; if the object
|
|
1030
|
+
# at this index is no longer that same object, the verdict no longer describes it
|
|
1031
|
+
# — discard this attempt and let the outer loop re-evaluate the index fresh rather
|
|
1032
|
+
# than apply a stale verdict to a new object.
|
|
1033
|
+
runtime = self._runtimes.get(world_index)
|
|
1034
|
+
if runtime is None or runtime is not probed_runtime:
|
|
1035
|
+
self._leased.discard(world_index)
|
|
1036
|
+
if runtime is not None:
|
|
1037
|
+
# The object was REPLACED, not removed -- put the index back where the
|
|
1038
|
+
# outer loop can find it, or it lands nowhere (not available, not down)
|
|
1039
|
+
# and every future lease() spins forever on a candidate set that never
|
|
1040
|
+
# grows.
|
|
1041
|
+
self._available.add(world_index)
|
|
1042
|
+
self._state_lock.notify_all()
|
|
1043
|
+
continue
|
|
1044
|
+
if is_healthy and runtime.state is RuntimeState.READY:
|
|
1045
|
+
self._state_lock.notify_all()
|
|
1046
|
+
return world_index, runtime
|
|
1047
|
+
cause = (
|
|
1048
|
+
f"reset failed: {reset_exc}"
|
|
1049
|
+
if reset_exc is not None
|
|
1050
|
+
else f"reset left world in state {runtime.state.value}"
|
|
1051
|
+
)
|
|
1052
|
+
# Preserve a typed §2f code (and its CARRIED domain, v1.15) across this seam
|
|
1053
|
+
# instead of flattening it to free text -- `mark_unhealthy` records it so a later
|
|
1054
|
+
# exhaustion declaration can tell a deterministic never-retried fault apart from a
|
|
1055
|
+
# generic infrastructure one.
|
|
1056
|
+
is_typed = isinstance(reset_exc, ProcessRuntimeError)
|
|
1057
|
+
code = reset_exc.code if is_typed else None
|
|
1058
|
+
domain = reset_exc.domain if is_typed else None
|
|
1059
|
+
|
|
1060
|
+
await self.mark_unhealthy(
|
|
1061
|
+
world_index, cause=cause, code=code, domain=domain
|
|
1062
|
+
)
|
|
1063
|
+
# loop again — this index is now excluded via `_down`, no explicit retry bookkeeping.
|
|
1064
|
+
|
|
1065
|
+
async def release(self, world_index: int) -> None:
|
|
1066
|
+
async with self._state_lock:
|
|
1067
|
+
self._leased.discard(world_index)
|
|
1068
|
+
if world_index in self._runtimes and world_index not in self._down:
|
|
1069
|
+
self._available.add(world_index)
|
|
1070
|
+
self._state_lock.notify_all()
|
|
1071
|
+
|
|
1072
|
+
async def mark_unhealthy(
|
|
1073
|
+
self,
|
|
1074
|
+
world_index: int,
|
|
1075
|
+
*,
|
|
1076
|
+
cause: str,
|
|
1077
|
+
code: str | None = None,
|
|
1078
|
+
domain: FailureDomain | None = None,
|
|
1079
|
+
) -> None:
|
|
1080
|
+
# `domain` is the CARRIED value off a typed provisioner error (v1.15); `None` here (e.g. a
|
|
1081
|
+
# caller that only has a bare code) falls back to the closed map via `_resolve_2f_domain`,
|
|
1082
|
+
# logged when it fires.
|
|
1083
|
+
async with self._state_lock:
|
|
1084
|
+
self._leased.discard(world_index)
|
|
1085
|
+
self._available.discard(world_index)
|
|
1086
|
+
self._fresh.discard(world_index)
|
|
1087
|
+
self._down.add(world_index)
|
|
1088
|
+
# Unconditional -- every demotion overwrites the recorded reason (or clears a stale
|
|
1089
|
+
# §2f code with `None` when this one isn't typed), so exhaustion always reads the
|
|
1090
|
+
# MOST RECENT cause for this index, never a leftover from an earlier failure.
|
|
1091
|
+
self._down_codes[world_index] = _resolve_2f_domain(code, domain)
|
|
1092
|
+
runtime = self._runtimes.get(world_index)
|
|
1093
|
+
if runtime is not None:
|
|
1094
|
+
# M12 (spine v1.12 §4.5b, normative): the scheduler demotes `state` on the
|
|
1095
|
+
# provider's own live `EnvironmentRuntime` object — that demotion is the signal
|
|
1096
|
+
# the NEXT `provision()` reconciles on.
|
|
1097
|
+
runtime.state = RuntimeState.UNHEALTHY
|
|
1098
|
+
# R10: set inside this same critical section (not left to `_schedule_reconcile`'s own,
|
|
1099
|
+
# later one) so a `lease()` observing state in the gap between the two never sees
|
|
1100
|
+
# "every world down, no reconcile in flight or pending" and raises spuriously.
|
|
1101
|
+
self._reconcile_pending = True
|
|
1102
|
+
self._state_lock.notify_all()
|
|
1103
|
+
|
|
1104
|
+
# Schedule recovery BEFORE the telemetry emit below -- `OutboundPort` calls are
|
|
1105
|
+
# best-effort and may be slow or hang, and recovery must never sit behind one (worst
|
|
1106
|
+
# case: `_reconcile_pending` stays latched and `lease()`'s grace loop spins forever).
|
|
1107
|
+
await self._schedule_reconcile()
|
|
1108
|
+
|
|
1109
|
+
# R6: this is the sole path every demotion (this method) goes through, so it is the one
|
|
1110
|
+
# place `world_unhealthy` needs to be emitted from for all four call sites to get it.
|
|
1111
|
+
if self._outbound is not None:
|
|
1112
|
+
try:
|
|
1113
|
+
await self._outbound.world_unhealthy(
|
|
1114
|
+
world_index=world_index, cause=_sanitize_cause(cause)
|
|
1115
|
+
)
|
|
1116
|
+
except (
|
|
1117
|
+
_FATAL_OUTBOUND
|
|
1118
|
+
) as exc: # a fence stops the run -- never best-effort.
|
|
1119
|
+
self.mark_fenced(exc)
|
|
1120
|
+
except Exception as exc: # noqa: BLE001 - B3: outbound failures are never fatal.
|
|
1121
|
+
await self._log(f"world_unhealthy emit failed: {exc}")
|
|
1122
|
+
|
|
1123
|
+
async def _schedule_reconcile(self) -> None:
|
|
1124
|
+
async with self._state_lock:
|
|
1125
|
+
if self._closed:
|
|
1126
|
+
return # R5: never spawn new provider work once the pool has been closed.
|
|
1127
|
+
if self._reconcile_in_flight():
|
|
1128
|
+
# B1/M6: a demotion landing mid-reconcile is coalesced into a trailing pass
|
|
1129
|
+
# (`_reconcile_loop`) rather than a second concurrent `provision()` call.
|
|
1130
|
+
self._reconcile_pending = True
|
|
1131
|
+
return
|
|
1132
|
+
self._reconcile_task = asyncio.create_task(self._reconcile_loop())
|
|
1133
|
+
|
|
1134
|
+
async def _reconcile_loop(self) -> None:
|
|
1135
|
+
while True:
|
|
1136
|
+
async with self._state_lock:
|
|
1137
|
+
self._reconcile_pending = False
|
|
1138
|
+
await self._reconcile()
|
|
1139
|
+
async with self._state_lock:
|
|
1140
|
+
if not self._reconcile_pending:
|
|
1141
|
+
return
|
|
1142
|
+
|
|
1143
|
+
async def _reconcile(self) -> None:
|
|
1144
|
+
# M5: bounded retry with backoff — a single transient `provision()` failure (a momentary
|
|
1145
|
+
# ENOSPC, an engine hiccup) used to retire its world for the rest of the job with no
|
|
1146
|
+
# signal anywhere. Every failed attempt is logged through `OutboundPort` (when wired),
|
|
1147
|
+
# matching the contract's own "loud, never silent" standard for degradation.
|
|
1148
|
+
runtimes: list[EnvironmentRuntime] | None = None
|
|
1149
|
+
last_exc: Exception | None = None
|
|
1150
|
+
for attempt in range(1, _RECONCILE_MAX_ATTEMPTS + 1):
|
|
1151
|
+
if self._closing:
|
|
1152
|
+
# R4: close() is already bounded-waiting on this task — do not spend its wait
|
|
1153
|
+
# budget retrying a pool that is being torn down anyway.
|
|
1154
|
+
return
|
|
1155
|
+
try:
|
|
1156
|
+
async with self._provider_lock:
|
|
1157
|
+
runtimes = await self._provisioner.provision(
|
|
1158
|
+
self._bundle,
|
|
1159
|
+
source=self._source,
|
|
1160
|
+
bundle_dir=self._bundle_dir,
|
|
1161
|
+
work_directory=self._work_directory,
|
|
1162
|
+
instances=self._instances,
|
|
1163
|
+
)
|
|
1164
|
+
except Exception as exc: # noqa: BLE001 - a reconcile must never crash the pool
|
|
1165
|
+
last_exc = exc
|
|
1166
|
+
await self._log(
|
|
1167
|
+
f"world pool reconcile attempt {attempt}/{_RECONCILE_MAX_ATTEMPTS} failed: {exc}"
|
|
1168
|
+
)
|
|
1169
|
+
if attempt < _RECONCILE_MAX_ATTEMPTS and not self._closing:
|
|
1170
|
+
await asyncio.sleep(_RECONCILE_BACKOFF_SECONDS[attempt - 1])
|
|
1171
|
+
continue
|
|
1172
|
+
last_exc = None
|
|
1173
|
+
break
|
|
1174
|
+
|
|
1175
|
+
if last_exc is not None or runtimes is None:
|
|
1176
|
+
# The FINAL failed re-provision attempt's typed §2f code (and its CARRIED domain,
|
|
1177
|
+
# v1.15), applied to every world still down when this reconcile gives up -- one
|
|
1178
|
+
# `provision()` call covers the whole pool, so a typed failure here is uniform by
|
|
1179
|
+
# construction across everything it did not just recover.
|
|
1180
|
+
is_typed = isinstance(last_exc, ProcessRuntimeError)
|
|
1181
|
+
code_and_domain = _resolve_2f_domain(
|
|
1182
|
+
last_exc.code if is_typed else None,
|
|
1183
|
+
last_exc.domain if is_typed else None,
|
|
1184
|
+
)
|
|
1185
|
+
# R8: every success path below ends in `notify_all()` — this give-up path must too,
|
|
1186
|
+
# or a `lease()` blocked in `_wait_bounded(poll=False)` (the `abandon is None` case)
|
|
1187
|
+
# waits forever for a reconcile that already gave up.
|
|
1188
|
+
# Unconditional, mirroring `mark_unhealthy`'s own invariant -- an untyped final
|
|
1189
|
+
# attempt must overwrite (clear) a stale typed code left by an earlier demotion, or
|
|
1190
|
+
# exhaustion later reads that leftover code as if it were this attempt's own result.
|
|
1191
|
+
async with self._state_lock:
|
|
1192
|
+
for index in self._down:
|
|
1193
|
+
self._down_codes[index] = code_and_domain
|
|
1194
|
+
self._state_lock.notify_all()
|
|
1195
|
+
return # stays `_down`; the next `mark_unhealthy` (or a lease-triggered wait) retries.
|
|
1196
|
+
|
|
1197
|
+
# M12: recovery is judged by re-probing `healthy()` (M2's port), never by reading `state`
|
|
1198
|
+
# back — the scheduler is what wrote `state` when it demoted this world, so trusting it
|
|
1199
|
+
# here would be reading our own signal as independent proof. R13 (spine v1.12 §4.5b):
|
|
1200
|
+
# `healthy` now rides the port's non-reentrancy rule, so these probes go under
|
|
1201
|
+
# `_provider_lock` too.
|
|
1202
|
+
if self._closing:
|
|
1203
|
+
# provision() just succeeded, but close() may already be queued on `_provider_lock`
|
|
1204
|
+
# for its own `provisioner.close()` call -- bail before racing it for one more round
|
|
1205
|
+
# of provider calls the pool is being torn down under anyway.
|
|
1206
|
+
return
|
|
1207
|
+
healthy_by_index: dict[int, bool] = {}
|
|
1208
|
+
# The probe's own §2f code, carried alongside its verdict -- a world that comes back from a
|
|
1209
|
+
# SUCCESSFUL `provision()` but fails this probe never enters the give-up path above (that
|
|
1210
|
+
# path only fires on a raised/failed `provision()`), so without this the state block below
|
|
1211
|
+
# has no code of its own and would otherwise leave whatever an earlier, superseded demotion
|
|
1212
|
+
# recorded standing.
|
|
1213
|
+
healthy_codes: dict[int, tuple[str, FailureDomain] | None] = {}
|
|
1214
|
+
async with self._provider_lock:
|
|
1215
|
+
for runtime in runtimes:
|
|
1216
|
+
try:
|
|
1217
|
+
healthy_by_index[
|
|
1218
|
+
runtime.world_index
|
|
1219
|
+
] = await self._provisioner.healthy(
|
|
1220
|
+
runtime, work_directory=self._work_directory
|
|
1221
|
+
)
|
|
1222
|
+
healthy_codes[runtime.world_index] = None
|
|
1223
|
+
except Exception as exc: # noqa: BLE001
|
|
1224
|
+
healthy_by_index[runtime.world_index] = False
|
|
1225
|
+
is_typed = isinstance(exc, ProcessRuntimeError)
|
|
1226
|
+
healthy_codes[runtime.world_index] = _resolve_2f_domain(
|
|
1227
|
+
exc.code if is_typed else None,
|
|
1228
|
+
exc.domain if is_typed else None,
|
|
1229
|
+
)
|
|
1230
|
+
|
|
1231
|
+
achieved = {runtime.world_index for runtime in runtimes}
|
|
1232
|
+
async with self._state_lock:
|
|
1233
|
+
for runtime in runtimes:
|
|
1234
|
+
self._runtimes[runtime.world_index] = runtime
|
|
1235
|
+
if healthy_by_index.get(runtime.world_index, False):
|
|
1236
|
+
was_down = runtime.world_index in self._down
|
|
1237
|
+
self._down.discard(runtime.world_index)
|
|
1238
|
+
self._down_codes.pop(
|
|
1239
|
+
runtime.world_index, None
|
|
1240
|
+
) # recovered -- stale now
|
|
1241
|
+
if runtime.world_index not in self._leased:
|
|
1242
|
+
self._available.add(runtime.world_index)
|
|
1243
|
+
if was_down and runtime.state is RuntimeState.READY:
|
|
1244
|
+
self._fresh.add(runtime.world_index) # m9
|
|
1245
|
+
elif runtime.world_index in self._down:
|
|
1246
|
+
# Still down after a successful re-provision -- this probe's own result
|
|
1247
|
+
# replaces whatever an earlier demotion left, never a leftover from before it.
|
|
1248
|
+
self._down_codes[runtime.world_index] = healthy_codes.get(
|
|
1249
|
+
runtime.world_index
|
|
1250
|
+
)
|
|
1251
|
+
# `provision` reconciles to exactly `instances` worlds (a conformance-gate degrade can
|
|
1252
|
+
# shrink `achieved` below what this pool started with) — anything no longer returned
|
|
1253
|
+
# is gone, not merely unhealthy.
|
|
1254
|
+
for stale in [index for index in self._runtimes if index not in achieved]:
|
|
1255
|
+
self._runtimes.pop(stale, None)
|
|
1256
|
+
self._available.discard(stale)
|
|
1257
|
+
self._down.discard(stale)
|
|
1258
|
+
self._fresh.discard(stale)
|
|
1259
|
+
self._down_codes.pop(stale, None) # the index itself is gone
|
|
1260
|
+
# m3: NOT `_leased.discard(stale)` — an in-flight scenario may still hold this
|
|
1261
|
+
# index's lease (e.g. a conformance degrade shrinking `achieved` mid-scenario);
|
|
1262
|
+
# dropping the lease record here would make its later `release()`/
|
|
1263
|
+
# `mark_unhealthy()` a silent no-op. Those methods already guard on
|
|
1264
|
+
# `world_index in self._runtimes`, so leaving `_leased` alone and letting them
|
|
1265
|
+
# reconcile it lazily is correct.
|
|
1266
|
+
# Keep this truthful across a reconcile, not just at start() -- P10 sizes
|
|
1267
|
+
# `parallelism_degraded` off it, and a reconcile can grow the pool back up or shrink
|
|
1268
|
+
# it further (a conformance degrade narrowing `achieved`) in either direction.
|
|
1269
|
+
self._effective_size = len(self._runtimes)
|
|
1270
|
+
self._state_lock.notify_all()
|
|
1271
|
+
|
|
1272
|
+
async def close(self) -> None:
|
|
1273
|
+
async with self._state_lock:
|
|
1274
|
+
if not self._closed:
|
|
1275
|
+
self._closed = True
|
|
1276
|
+
self._closing = True
|
|
1277
|
+
# Wake anything blocked in `lease()`'s `_wait_bounded(poll=False)` so it
|
|
1278
|
+
# re-checks `_closed` instead of waiting for a recovery that will never come.
|
|
1279
|
+
self._state_lock.notify_all()
|
|
1280
|
+
# The OLD idempotency check (`if self._closed: return`) latched here, before teardown
|
|
1281
|
+
# ever ran -- a caller wrapping this whole call in its own timeout (the entrypoint's
|
|
1282
|
+
# `_bounded_close`) could cancel it mid-teardown, and a RETRY then hit that early
|
|
1283
|
+
# return and silently never called `provisioner.close()` at all. `_closed` still has
|
|
1284
|
+
# to latch immediately (lease()'s top-of-loop check, its inner re-check under
|
|
1285
|
+
# `_provider_lock`, and `_teardown`'s own bail-out below all depend on new
|
|
1286
|
+
# leases/reconciles being rejected the moment close() STARTS, not once it finishes),
|
|
1287
|
+
# so idempotency now lives on a separate, SHARED teardown task instead: every call —
|
|
1288
|
+
# first or retried — creates it once and awaits the same one.
|
|
1289
|
+
if self._teardown_task is None:
|
|
1290
|
+
self._teardown_task = asyncio.create_task(self._teardown())
|
|
1291
|
+
teardown_task = self._teardown_task
|
|
1292
|
+
|
|
1293
|
+
# `asyncio.shield`: if THIS call's own awaiter is cancelled (the caller's timeout fires),
|
|
1294
|
+
# the cancellation stops at this `await` and never reaches `teardown_task` -- teardown
|
|
1295
|
+
# keeps running in the background, and a retry's `close()` re-attaches to the same
|
|
1296
|
+
# (possibly by-then-finished) task instead of no-op'ing.
|
|
1297
|
+
await asyncio.shield(teardown_task)
|
|
1298
|
+
|
|
1299
|
+
async def _teardown(self) -> None:
|
|
1300
|
+
async with self._state_lock:
|
|
1301
|
+
task = self._reconcile_task
|
|
1302
|
+
if task is not None:
|
|
1303
|
+
# §4.5b: do NOT cancel-then-close.
|
|
1304
|
+
# `ProcessRuntimeProvider.provision`/`reset`/`healthy` are `asyncio.to_thread` —
|
|
1305
|
+
# cancelling the awaiting coroutine does NOT stop the underlying thread, so the old
|
|
1306
|
+
# bounded-wait-then-cancel let `provisioner.close()` run CONCURRENTLY with a
|
|
1307
|
+
# still-live `provision()` once the bound expired: unsynchronized identity dicts
|
|
1308
|
+
# (`RuntimeError: dictionary changed size during iteration`), leaked engines, and
|
|
1309
|
+
# `close()` itself could raise out of the guest's terminal path. `_closing` (in
|
|
1310
|
+
# `_reconcile`'s own retry loop and healthy-probe gate) already makes a reconcile bail
|
|
1311
|
+
# BETWEEN attempts/probes without a cancel, so this waits for the ONE call already in
|
|
1312
|
+
# flight to finish on its own — unbounded from this function's perspective, but
|
|
1313
|
+
# bounded in practice by whichever single `provision()`/`healthy()` call was running,
|
|
1314
|
+
# with the outer flush-window deadline (spine, P10-owned) as the real backstop --
|
|
1315
|
+
# there is no longer a single constant here that bounds this wait on its own.
|
|
1316
|
+
await asyncio.gather(task, return_exceptions=True)
|
|
1317
|
+
|
|
1318
|
+
async with self._provider_lock:
|
|
1319
|
+
await self._provisioner.close(work_directory=self._work_directory)
|
|
1320
|
+
|
|
1321
|
+
async def _log(self, message: str, *, level: str = "error") -> None:
|
|
1322
|
+
if self._outbound is None:
|
|
1323
|
+
return
|
|
1324
|
+
try:
|
|
1325
|
+
# R9: reuses `world_unhealthy.cause`'s own sanitizer — a `provision()` failure
|
|
1326
|
+
# routinely carries a postgres error string with the DSN, and outbound-channels.md
|
|
1327
|
+
# requires redaction (no endpoint userinfo) before anything crosses the wire.
|
|
1328
|
+
await self._outbound.log(level=level, message=_sanitize_cause(message))
|
|
1329
|
+
except _FATAL_OUTBOUND as exc: # a fence stops the run -- never best-effort.
|
|
1330
|
+
self.mark_fenced(exc)
|
|
1331
|
+
except Exception: # noqa: BLE001 - B3: outbound failures are never fatal.
|
|
1332
|
+
pass
|
|
1333
|
+
|
|
1334
|
+
|
|
1335
|
+
# --- the scenario loop --------------------------------------------------------------------------
|
|
1336
|
+
|
|
1337
|
+
|
|
1338
|
+
@dataclass(frozen=True)
|
|
1339
|
+
class RunResult:
|
|
1340
|
+
"""`receipts` mixes already-emitted real receipts with synthesized-but-not-yet-emitted
|
|
1341
|
+
`skipped` ones (R6) — see `HostedScheduler.emit_skipped_receipts`.
|
|
1342
|
+
|
|
1343
|
+
`fenced` is set once a 401/403 (`HostedFencedError`), a 404-exhausted channel
|
|
1344
|
+
(`HostedChannelFailedError`), or a 409 attempt-supersession (`HostedAttemptSupersededError`)
|
|
1345
|
+
was observed on any outbound call -- the run stops launching further scenarios the moment it is
|
|
1346
|
+
set. The caller maps this to exit code 3 and must not call `emit_skipped_receipts` (no further
|
|
1347
|
+
outbound emission once fenced)."""
|
|
1348
|
+
|
|
1349
|
+
receipts: tuple[ResultReceipt, ...]
|
|
1350
|
+
aborted: ReceiptFailure | None
|
|
1351
|
+
fenced: BaseException | None = None
|
|
1352
|
+
|
|
1353
|
+
|
|
1354
|
+
def _skipped_receipt(scenario: Scenario) -> ResultReceipt:
|
|
1355
|
+
# Exact body per outbound-channels.md Channel 2, "skipped receipt body (exact)".
|
|
1356
|
+
return ResultReceipt(
|
|
1357
|
+
scenario_key=scenario.scenario_key,
|
|
1358
|
+
scenario_id=scenario.scenario_id,
|
|
1359
|
+
scenario_attempt=1,
|
|
1360
|
+
world_index=None,
|
|
1361
|
+
status="skipped",
|
|
1362
|
+
sub_goals=(),
|
|
1363
|
+
evaluations=(),
|
|
1364
|
+
call=None,
|
|
1365
|
+
failure=None,
|
|
1366
|
+
)
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
def _unjudged(sub_goals: Sequence[SubGoal]) -> tuple[SubGoalResult, ...]:
|
|
1370
|
+
return tuple(
|
|
1371
|
+
# R11: outbound-channels.md pins `judged` as `SubGoal.judged != ""`, not `bool(...)` —
|
|
1372
|
+
# they agree for every `str` but `bool` is not what the contract names.
|
|
1373
|
+
SubGoalResult(name=goal.name, held=None, reason=None, judged=goal.judged != "")
|
|
1374
|
+
for goal in sub_goals
|
|
1375
|
+
)
|
|
1376
|
+
|
|
1377
|
+
|
|
1378
|
+
_LEAK_HEADROOM = (
|
|
1379
|
+
10 # R1: spine §1's hosted `scenario_count` admission range is 1..10 -- the most
|
|
1380
|
+
)
|
|
1381
|
+
# phase threads that can ever be simultaneously abandoned (leaked) in one job.
|
|
1382
|
+
|
|
1383
|
+
|
|
1384
|
+
def _abort_from_no_worlds(exc: NoWorldsAvailable) -> ReceiptFailure:
|
|
1385
|
+
# A uniform §2f never-retried code across every unhealthy world surfaces AS that code+domain;
|
|
1386
|
+
# otherwise this is the generic exhaustion abort.
|
|
1387
|
+
if exc.code is not None and exc.domain is not None:
|
|
1388
|
+
return ReceiptFailure(
|
|
1389
|
+
domain=exc.domain.value,
|
|
1390
|
+
stage=HarnessStage.RUNNING.value,
|
|
1391
|
+
code=exc.code,
|
|
1392
|
+
message=_truncate(str(exc)),
|
|
1393
|
+
)
|
|
1394
|
+
return _failure("world_pool_exhausted", str(exc))
|
|
1395
|
+
|
|
1396
|
+
|
|
1397
|
+
@dataclass
|
|
1398
|
+
class _ScenarioContext:
|
|
1399
|
+
"""R7: `_run_scenario` records the world/attempt it is currently working on here as it goes,
|
|
1400
|
+
so a crash that escapes every handled path still lets `worker()` report the REAL
|
|
1401
|
+
world_index/scenario_attempt on the `driver_crashed` receipt instead of always None/1.
|
|
1402
|
+
`call` is set the moment the call step returns, so a LATER crash (e.g. `read_only()`
|
|
1403
|
+
building the check-phase handle) still reports the call that genuinely ran, not `null`."""
|
|
1404
|
+
|
|
1405
|
+
world_index: int | None = None
|
|
1406
|
+
attempt: int = 1
|
|
1407
|
+
call: CallSummary | None = None
|
|
1408
|
+
|
|
1409
|
+
|
|
1410
|
+
@dataclass(frozen=True)
|
|
1411
|
+
class _PendingRetryReceipt:
|
|
1412
|
+
"""R3: carries attempt-1's already-built `_Retry` outcome across the retry-lease boundary, so
|
|
1413
|
+
a cancel/abort landing anywhere between "attempt 1 finished" and "attempt 2 actually starts"
|
|
1414
|
+
still reports what attempt 1 produced instead of losing it to skipped-synthesis (the same
|
|
1415
|
+
defect M8 fixed on the `NoWorldsAvailable` branch, on the other post-attempt-1 exit)."""
|
|
1416
|
+
|
|
1417
|
+
world_index: int
|
|
1418
|
+
attempt: int
|
|
1419
|
+
outcome: "_Retry"
|
|
1420
|
+
|
|
1421
|
+
|
|
1422
|
+
def _record_scenario(span: Any, receipt: Any, context: Any) -> None:
|
|
1423
|
+
"""Put the scenario's verdict on its span, so a trace answers what happened without a receipt."""
|
|
1424
|
+
if span is None or receipt is None:
|
|
1425
|
+
return
|
|
1426
|
+
failure = getattr(receipt, "failure", None)
|
|
1427
|
+
observability.record(
|
|
1428
|
+
span,
|
|
1429
|
+
status=getattr(receipt, "status", None),
|
|
1430
|
+
failure_code=getattr(failure, "code", None),
|
|
1431
|
+
failure_domain=getattr(failure, "domain", None),
|
|
1432
|
+
world_index=getattr(context, "world_index", None),
|
|
1433
|
+
attempt=getattr(context, "attempt", None),
|
|
1434
|
+
)
|
|
1435
|
+
|
|
1436
|
+
class HostedScheduler:
|
|
1437
|
+
"""Drains a job's scenario list across a `WorldPool`, one asyncio task per scenario — lease()
|
|
1438
|
+
blocking when the pool is saturated is what caps concurrency at W, so nothing here re-derives
|
|
1439
|
+
a worker count. Retry is fixed at one extra attempt on a fresh world (spine §5 step 4), gated
|
|
1440
|
+
on `FailureDomain` per the P9 brief: retryable domains retry once, deterministic ones do not.
|
|
1441
|
+
"""
|
|
1442
|
+
|
|
1443
|
+
def __init__(
|
|
1444
|
+
self,
|
|
1445
|
+
*,
|
|
1446
|
+
pool: WorldPool,
|
|
1447
|
+
world_factory: WorldFactory,
|
|
1448
|
+
call_runner: CallRunner,
|
|
1449
|
+
outbound: OutboundPort,
|
|
1450
|
+
job_seed: int,
|
|
1451
|
+
cancel_requested: Callable[[], bool] | None = None,
|
|
1452
|
+
judge: JudgeFn | None = None,
|
|
1453
|
+
) -> None:
|
|
1454
|
+
self._pool = pool
|
|
1455
|
+
self._world_factory = world_factory
|
|
1456
|
+
self._call_runner = call_runner
|
|
1457
|
+
self._outbound = outbound
|
|
1458
|
+
self._job_seed = job_seed
|
|
1459
|
+
self._cancel_requested = cancel_requested or (lambda: False)
|
|
1460
|
+
# Injected like every other collaborator, so a test decides a judged sub-goal without a
|
|
1461
|
+
# model call. Resolved here rather than as a default argument, which would bind at import
|
|
1462
|
+
# and ignore both injection and patching.
|
|
1463
|
+
self._judge = judge or _judge
|
|
1464
|
+
self._executor: ThreadPoolExecutor | None = None
|
|
1465
|
+
|
|
1466
|
+
async def run(self, scenarios: Sequence[Scenario]) -> RunResult:
|
|
1467
|
+
results: list[ResultReceipt | None] = [None] * len(scenarios)
|
|
1468
|
+
abort_holder: list[ReceiptFailure | None] = [None]
|
|
1469
|
+
|
|
1470
|
+
# R1: a dedicated executor for scenario phase threads — never the loop's default
|
|
1471
|
+
# executor, which the provider's own `to_thread` calls (process_runtime.py) also use, and
|
|
1472
|
+
# whose capacity a leaked phase thread would starve globally. One worker per live world
|
|
1473
|
+
# plus headroom for the worst case of every admitted scenario leaking its own abandoned
|
|
1474
|
+
# thread at once (world-handle-interface.md: "its thread leaks, bounded by scenario
|
|
1475
|
+
# count").
|
|
1476
|
+
self._executor = ThreadPoolExecutor(
|
|
1477
|
+
# `_LEAK_HEADROOM` alone assumes spine §1's `scenario_count` admission cap (<=10) —
|
|
1478
|
+
# widen for whatever `scenarios` actually holds, or an over-cap job's overflow
|
|
1479
|
+
# scenarios find the executor saturated and report `driver_crashed` for a phase that
|
|
1480
|
+
# was queued, not run.
|
|
1481
|
+
max_workers=max(
|
|
1482
|
+
self._pool.effective_size + _LEAK_HEADROOM, len(scenarios) + 1
|
|
1483
|
+
),
|
|
1484
|
+
thread_name_prefix="hosted-scenario",
|
|
1485
|
+
)
|
|
1486
|
+
try:
|
|
1487
|
+
|
|
1488
|
+
async def worker(index: int, scenario: Scenario) -> None:
|
|
1489
|
+
# `self._pool.fenced` is the same stop-path as `abort_holder`/`cancel_requested`
|
|
1490
|
+
# -- once any outbound call has hit a 401/403 or an exhausted channel, no further
|
|
1491
|
+
# scenario may even start.
|
|
1492
|
+
if (
|
|
1493
|
+
abort_holder[0] is not None
|
|
1494
|
+
or self._pool.fenced is not None
|
|
1495
|
+
or self._cancel_requested()
|
|
1496
|
+
):
|
|
1497
|
+
return
|
|
1498
|
+
context = _ScenarioContext()
|
|
1499
|
+
try:
|
|
1500
|
+
with observability.scenario(
|
|
1501
|
+
str(getattr(scenario, "key", "") or index), index
|
|
1502
|
+
) as span:
|
|
1503
|
+
results[index] = await self._run_scenario(
|
|
1504
|
+
scenario, index, abort_holder=abort_holder, context=context
|
|
1505
|
+
)
|
|
1506
|
+
_record_scenario(span, results[index], context)
|
|
1507
|
+
except NoWorldsAvailable as exc:
|
|
1508
|
+
abort_holder[0] = _abort_from_no_worlds(exc)
|
|
1509
|
+
except _FATAL_OUTBOUND:
|
|
1510
|
+
# Already latched onto `self._pool.fenced` by whichever `_emit`/`_log` call
|
|
1511
|
+
# raised it -- no receipt for a scenario the platform already superseded.
|
|
1512
|
+
pass
|
|
1513
|
+
except asyncio.CancelledError:
|
|
1514
|
+
raise
|
|
1515
|
+
except BaseException as exc: # noqa: BLE001
|
|
1516
|
+
# B3: the scheduler's own machinery crashing must not suppress every other
|
|
1517
|
+
# scenario's receipt — `gather(return_exceptions=True)` below is the second
|
|
1518
|
+
# half of that guarantee.
|
|
1519
|
+
results[index] = await self._driver_crashed_receipt(
|
|
1520
|
+
scenario,
|
|
1521
|
+
exc,
|
|
1522
|
+
world_index=context.world_index,
|
|
1523
|
+
scenario_attempt=context.attempt,
|
|
1524
|
+
call=context.call,
|
|
1525
|
+
)
|
|
1526
|
+
|
|
1527
|
+
tasks = [asyncio.create_task(worker(i, s)) for i, s in enumerate(scenarios)]
|
|
1528
|
+
if tasks:
|
|
1529
|
+
await asyncio.gather(*tasks, return_exceptions=True)
|
|
1530
|
+
|
|
1531
|
+
receipts: list[ResultReceipt] = []
|
|
1532
|
+
for index, scenario in enumerate(scenarios):
|
|
1533
|
+
receipt = results[index]
|
|
1534
|
+
if receipt is None:
|
|
1535
|
+
# (outbound-channels.md v1.3 Sequencing: "terminal event -> skipped
|
|
1536
|
+
# receipts -> manifest"): only SYNTHESIZE here. `run()` returns before its
|
|
1537
|
+
# caller has emitted a terminal event, so pushing this over `outbound` now
|
|
1538
|
+
# would put it on the wire ahead of the terminal -- `emit_skipped_receipts()`
|
|
1539
|
+
# is the caller's job, done AFTER its own terminal event.
|
|
1540
|
+
receipt = _skipped_receipt(scenario)
|
|
1541
|
+
receipts.append(receipt)
|
|
1542
|
+
return RunResult(
|
|
1543
|
+
receipts=tuple(receipts),
|
|
1544
|
+
aborted=abort_holder[0],
|
|
1545
|
+
fenced=self._pool.fenced,
|
|
1546
|
+
)
|
|
1547
|
+
finally:
|
|
1548
|
+
# R1: never block `run()` on abandoned threads — `shutdown(wait=True)` would hang
|
|
1549
|
+
# this coroutine exactly like the bug this fixes. Queued-but-unstarted work is
|
|
1550
|
+
# cancelled; already-running (leaked) threads are the contract's own accepted,
|
|
1551
|
+
# bounded tradeoff (world-handle-interface.md's "the job TTL is the backstop").
|
|
1552
|
+
self._executor.shutdown(wait=False, cancel_futures=True)
|
|
1553
|
+
|
|
1554
|
+
async def emit_skipped_receipts(self, result: RunResult) -> None:
|
|
1555
|
+
"""R6: outbound-channels.md v1.3 Sequencing — "terminal event -> skipped receipts ->
|
|
1556
|
+
manifest". `run()` only synthesizes `skipped` receipts into `RunResult.receipts`; call
|
|
1557
|
+
this AFTER the caller's own terminal event has been emitted, never before, and exactly
|
|
1558
|
+
once — each call re-emits every `skipped` receipt in `result.receipts` with no dedup of
|
|
1559
|
+
its own.
|
|
1560
|
+
|
|
1561
|
+
A no-op once `self._pool.fenced` is set (checked live, so it also covers a fence that
|
|
1562
|
+
landed after `run()` returned but before this call) — the run stopped emitting the moment
|
|
1563
|
+
the fence was observed and must not resume for these. If a fence instead lands DURING this
|
|
1564
|
+
method's own loop, the same `_FATAL_OUTBOUND` that stops `run()` escapes out of this method
|
|
1565
|
+
too; the caller must be ready for that."""
|
|
1566
|
+
if self._pool.fenced is not None:
|
|
1567
|
+
return
|
|
1568
|
+
for receipt in result.receipts:
|
|
1569
|
+
if receipt.status == "skipped":
|
|
1570
|
+
await self._emit(self._outbound.receipt(receipt), what="receipt")
|
|
1571
|
+
|
|
1572
|
+
async def _emit(self, awaitable: Awaitable[None], *, what: str) -> None:
|
|
1573
|
+
# B3: `OutboundPort` exceptions are best-effort telemetry — never receipt-affecting and
|
|
1574
|
+
# never fatal to the run. Logged through the same port when logging itself doesn't also
|
|
1575
|
+
# fail; swallowed otherwise rather than let a transport hiccup kill the scenario loop.
|
|
1576
|
+
# The one exception besides `CancelledError` this deliberately does NOT swallow — a
|
|
1577
|
+
# fence (401/403) or an exhausted channel (404x3) is never best-effort telemetry.
|
|
1578
|
+
try:
|
|
1579
|
+
await awaitable
|
|
1580
|
+
except _FATAL_OUTBOUND as exc:
|
|
1581
|
+
self._pool.mark_fenced(exc)
|
|
1582
|
+
raise
|
|
1583
|
+
except Exception as exc: # noqa: BLE001
|
|
1584
|
+
try:
|
|
1585
|
+
await self._outbound.log(
|
|
1586
|
+
level="error", message=f"outbound.{what} failed: {exc}"
|
|
1587
|
+
)
|
|
1588
|
+
except _FATAL_OUTBOUND as log_exc:
|
|
1589
|
+
self._pool.mark_fenced(log_exc)
|
|
1590
|
+
raise
|
|
1591
|
+
except Exception: # noqa: BLE001
|
|
1592
|
+
pass
|
|
1593
|
+
|
|
1594
|
+
async def _driver_crashed_receipt(
|
|
1595
|
+
self,
|
|
1596
|
+
scenario: Scenario,
|
|
1597
|
+
exc: BaseException,
|
|
1598
|
+
*,
|
|
1599
|
+
world_index: int | None,
|
|
1600
|
+
scenario_attempt: int,
|
|
1601
|
+
call: CallSummary | None = None,
|
|
1602
|
+
) -> ResultReceipt:
|
|
1603
|
+
failure = _failure("driver_crashed", f"{type(exc).__name__}: {exc}")
|
|
1604
|
+
try:
|
|
1605
|
+
# R7: best-effort — every declared goal, `held: null`, matching the errored-receipt
|
|
1606
|
+
# body's rule. Falls back to `()` only when reading `sub_goals` itself is what crashed
|
|
1607
|
+
# (the one case with no goal list to report at all).
|
|
1608
|
+
sub_goals = _unjudged(scenario.sub_goals)
|
|
1609
|
+
except Exception: # noqa: BLE001
|
|
1610
|
+
sub_goals = ()
|
|
1611
|
+
receipt = ResultReceipt(
|
|
1612
|
+
scenario_key=scenario.scenario_key,
|
|
1613
|
+
scenario_id=scenario.scenario_id,
|
|
1614
|
+
scenario_attempt=scenario_attempt,
|
|
1615
|
+
world_index=world_index,
|
|
1616
|
+
status="errored",
|
|
1617
|
+
sub_goals=sub_goals,
|
|
1618
|
+
evaluations=(),
|
|
1619
|
+
call=call, # the call step's own summary, if it had already returned when this crashed
|
|
1620
|
+
failure=failure,
|
|
1621
|
+
)
|
|
1622
|
+
await self._emit(self._outbound.receipt(receipt), what="receipt")
|
|
1623
|
+
return receipt
|
|
1624
|
+
|
|
1625
|
+
async def _emit_pending_retry_receipt(
|
|
1626
|
+
self, scenario: Scenario, pending: "_PendingRetryReceipt"
|
|
1627
|
+
) -> ResultReceipt:
|
|
1628
|
+
# R3: the single shape both post-attempt-1 "never got to run attempt 2" exits emit.
|
|
1629
|
+
receipt = ResultReceipt(
|
|
1630
|
+
scenario_key=scenario.scenario_key,
|
|
1631
|
+
scenario_id=scenario.scenario_id,
|
|
1632
|
+
scenario_attempt=pending.attempt,
|
|
1633
|
+
world_index=pending.world_index,
|
|
1634
|
+
status="errored",
|
|
1635
|
+
sub_goals=pending.outcome.sub_goals,
|
|
1636
|
+
evaluations=(),
|
|
1637
|
+
call=pending.outcome.call,
|
|
1638
|
+
failure=pending.outcome.failure,
|
|
1639
|
+
)
|
|
1640
|
+
await self._emit(self._outbound.receipt(receipt), what="receipt")
|
|
1641
|
+
return receipt
|
|
1642
|
+
|
|
1643
|
+
async def _lease_or_abandon(
|
|
1644
|
+
self, *, exclude: frozenset[int], abort_holder: list[ReceiptFailure | None]
|
|
1645
|
+
) -> tuple[int, EnvironmentRuntime] | None:
|
|
1646
|
+
def _abandon() -> bool:
|
|
1647
|
+
# A scenario already queued in `lease()` must also abandon once fenced -- the
|
|
1648
|
+
# worker-top check alone only stops scenarios that had not started yet.
|
|
1649
|
+
return (
|
|
1650
|
+
abort_holder[0] is not None
|
|
1651
|
+
or self._pool.fenced is not None
|
|
1652
|
+
or self._cancel_requested()
|
|
1653
|
+
)
|
|
1654
|
+
|
|
1655
|
+
return await self._pool.lease(exclude=exclude, abandon=_abandon)
|
|
1656
|
+
|
|
1657
|
+
async def _run_scenario(
|
|
1658
|
+
self,
|
|
1659
|
+
scenario: Scenario,
|
|
1660
|
+
scenario_index: int,
|
|
1661
|
+
*,
|
|
1662
|
+
attempt: int = 1,
|
|
1663
|
+
tried: frozenset[int] = frozenset(),
|
|
1664
|
+
pre_leased: tuple[int, EnvironmentRuntime] | None = None,
|
|
1665
|
+
pending_retry: "_PendingRetryReceipt | None" = None,
|
|
1666
|
+
abort_holder: list[ReceiptFailure | None],
|
|
1667
|
+
context: "_ScenarioContext",
|
|
1668
|
+
) -> ResultReceipt | None:
|
|
1669
|
+
if pre_leased is not None:
|
|
1670
|
+
world_index, runtime = pre_leased
|
|
1671
|
+
else:
|
|
1672
|
+
leased = await self._lease_or_abandon(
|
|
1673
|
+
exclude=tried, abort_holder=abort_holder
|
|
1674
|
+
)
|
|
1675
|
+
if leased is None:
|
|
1676
|
+
return None # B5: cancelled/aborted while queued — never got a world
|
|
1677
|
+
world_index, runtime = leased
|
|
1678
|
+
|
|
1679
|
+
context.world_index = (
|
|
1680
|
+
world_index # R7: the real values for a driver_crashed receipt
|
|
1681
|
+
)
|
|
1682
|
+
context.attempt = attempt
|
|
1683
|
+
context.call = (
|
|
1684
|
+
None # this attempt has not made its own call yet -- must not still
|
|
1685
|
+
)
|
|
1686
|
+
# carry a previous attempt's summary on the shared context object into this one's receipt.
|
|
1687
|
+
|
|
1688
|
+
# B5: re-check immediately after `lease()` returns — a cancel/abort landing while this
|
|
1689
|
+
# worker was queued must not let a freshly granted world start work it can never finish
|
|
1690
|
+
# inside the flush window.
|
|
1691
|
+
if abort_holder[0] is not None or self._cancel_requested():
|
|
1692
|
+
await self._pool.release(world_index)
|
|
1693
|
+
if pending_retry is not None:
|
|
1694
|
+
# R3: attempt 1 already ran on `pending_retry.world_index` and produced a real
|
|
1695
|
+
# outcome — this is the retry continuation (this world was never used for it).
|
|
1696
|
+
return await self._emit_pending_retry_receipt(scenario, pending_retry)
|
|
1697
|
+
return None
|
|
1698
|
+
|
|
1699
|
+
world_resolved = (
|
|
1700
|
+
False # B3: the leased world must be released/discarded exactly once
|
|
1701
|
+
)
|
|
1702
|
+
try:
|
|
1703
|
+
if pending_retry is not None:
|
|
1704
|
+
# Emitted here, immediately before attempt 2's own `scenario_started`, so this
|
|
1705
|
+
# event and the pending-retry receipt (both exits above) are mutually exclusive
|
|
1706
|
+
# by construction — outbound-channels.md Channel 2: "the failed first try is
|
|
1707
|
+
# recorded by scenario_retried/world_unhealthy events, never by a receipt."
|
|
1708
|
+
await self._emit(
|
|
1709
|
+
self._outbound.scenario_retried(
|
|
1710
|
+
scenario_key=scenario.scenario_key,
|
|
1711
|
+
from_world=pending_retry.world_index,
|
|
1712
|
+
to_world=world_index,
|
|
1713
|
+
),
|
|
1714
|
+
what="scenario_retried",
|
|
1715
|
+
)
|
|
1716
|
+
await self._emit(
|
|
1717
|
+
self._outbound.scenario_started(
|
|
1718
|
+
scenario_key=scenario.scenario_key,
|
|
1719
|
+
world_index=world_index,
|
|
1720
|
+
scenario_attempt=attempt,
|
|
1721
|
+
),
|
|
1722
|
+
what="scenario_started",
|
|
1723
|
+
)
|
|
1724
|
+
|
|
1725
|
+
rng = random.Random(self._job_seed + scenario_index)
|
|
1726
|
+
outcome: ResultReceipt | _Retry
|
|
1727
|
+
try:
|
|
1728
|
+
world = await self._world_factory.create(runtime, rng=rng)
|
|
1729
|
+
except Exception as exc: # noqa: BLE001
|
|
1730
|
+
# B3: `world_factory.create()` failing (e.g. a PostgresStore connect failure) is
|
|
1731
|
+
# the same shape as a mid-scenario `WorldUnavailable` — the world is unusable, not
|
|
1732
|
+
# the scenario code. Deliberately narrow to just this call: `_execute()` has its
|
|
1733
|
+
# own exhaustive internal exception handling (`_run_phase`/`_invoke`), so anything
|
|
1734
|
+
# that still escapes it is a genuine scheduler bug and belongs in `driver_crashed`
|
|
1735
|
+
# (via `worker()`'s `BaseException` catch), not swallowed into `world_unavailable`.
|
|
1736
|
+
outcome = _Retry(
|
|
1737
|
+
_failure("world_unavailable", f"{type(exc).__name__}: {exc}"),
|
|
1738
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1739
|
+
call=None,
|
|
1740
|
+
mark_unhealthy=True,
|
|
1741
|
+
)
|
|
1742
|
+
else:
|
|
1743
|
+
outcome = await self._execute(
|
|
1744
|
+
scenario,
|
|
1745
|
+
world,
|
|
1746
|
+
runtime,
|
|
1747
|
+
world_index,
|
|
1748
|
+
attempt=attempt,
|
|
1749
|
+
context=context,
|
|
1750
|
+
)
|
|
1751
|
+
|
|
1752
|
+
if isinstance(outcome, _Retry):
|
|
1753
|
+
# R6: `mark_unhealthy()` itself emits `world_unhealthy` now (every demotion path
|
|
1754
|
+
# goes through it) — no separate emit needed here.
|
|
1755
|
+
if outcome.mark_unhealthy:
|
|
1756
|
+
await self._pool.mark_unhealthy(
|
|
1757
|
+
world_index, cause=outcome.failure.message
|
|
1758
|
+
)
|
|
1759
|
+
else:
|
|
1760
|
+
await self._pool.release(world_index)
|
|
1761
|
+
world_resolved = True
|
|
1762
|
+
|
|
1763
|
+
if attempt >= 2:
|
|
1764
|
+
receipt = ResultReceipt(
|
|
1765
|
+
scenario_key=scenario.scenario_key,
|
|
1766
|
+
scenario_id=scenario.scenario_id,
|
|
1767
|
+
scenario_attempt=attempt,
|
|
1768
|
+
world_index=world_index,
|
|
1769
|
+
status="errored",
|
|
1770
|
+
sub_goals=outcome.sub_goals,
|
|
1771
|
+
evaluations=(),
|
|
1772
|
+
call=outcome.call,
|
|
1773
|
+
failure=outcome.failure,
|
|
1774
|
+
)
|
|
1775
|
+
await self._emit(self._outbound.receipt(receipt), what="receipt")
|
|
1776
|
+
return receipt
|
|
1777
|
+
|
|
1778
|
+
# R3: attempt 1's outcome, carried forward so either exit below that never gets to
|
|
1779
|
+
# start attempt 2 can still report it instead of losing it to skipped-synthesis.
|
|
1780
|
+
pending = _PendingRetryReceipt(
|
|
1781
|
+
world_index=world_index, attempt=attempt, outcome=outcome
|
|
1782
|
+
)
|
|
1783
|
+
# A retry normally moves to another world. With an effective one-world pool,
|
|
1784
|
+
# excluding the only world makes the promised retry impossible: lease() raises
|
|
1785
|
+
# NoWorldsAvailable even though release() has returned a healthy runtime and the
|
|
1786
|
+
# next lease will reset it. Reuse is safe for non-unhealthy failures because the
|
|
1787
|
+
# lease path always resets the world before attempt 2. An unhealthy world is
|
|
1788
|
+
# still demoted and therefore cannot be leased until reconciliation replaces it.
|
|
1789
|
+
retry_exclude = tried | {world_index}
|
|
1790
|
+
if self._pool.effective_size == 1:
|
|
1791
|
+
retry_exclude = frozenset()
|
|
1792
|
+
try:
|
|
1793
|
+
next_leased = await self._lease_or_abandon(
|
|
1794
|
+
exclude=retry_exclude, abort_holder=abort_holder
|
|
1795
|
+
)
|
|
1796
|
+
except NoWorldsAvailable as exc:
|
|
1797
|
+
# M8: this scenario already ran and produced a real attempt-1 failure — losing
|
|
1798
|
+
# it to skipped-synthesis just because the retry lease found nothing would
|
|
1799
|
+
# report "never ran" for a scenario that manifestly did.
|
|
1800
|
+
#
|
|
1801
|
+
# P=1 failure preservation (v1.15): when the pool exhaustion carries no typed
|
|
1802
|
+
# §2f code of its own (exc.code is None), the original call failure is more
|
|
1803
|
+
# informative than the generic world_pool_exhausted — preserve it as the
|
|
1804
|
+
# job-level abort so a deterministic call_failed/CallAborted is returned as
|
|
1805
|
+
# its original typed failure, not masked behind world_pool_exhausted.
|
|
1806
|
+
# The scenario already produced the causative typed failure. A subsequent
|
|
1807
|
+
# inability to allocate its retry world must not replace that evidence with a
|
|
1808
|
+
# generic pool/provisioning wrapper (the masking seen in the hosted voice
|
|
1809
|
+
# timeout run).
|
|
1810
|
+
if pending.outcome.failure is not None:
|
|
1811
|
+
abort_holder[0] = pending.outcome.failure
|
|
1812
|
+
else:
|
|
1813
|
+
abort_holder[0] = _abort_from_no_worlds(exc) # v1.13 §5.4
|
|
1814
|
+
return await self._emit_pending_retry_receipt(scenario, pending)
|
|
1815
|
+
|
|
1816
|
+
if next_leased is None:
|
|
1817
|
+
# R3: same defect as the branch above, reached via cancel/abort instead of
|
|
1818
|
+
# pool exhaustion. Preserve the attempt's concrete failure when pool repair
|
|
1819
|
+
# failed first and installed a generic infrastructure abort.
|
|
1820
|
+
if pending.outcome.failure is not None:
|
|
1821
|
+
abort_holder[0] = pending.outcome.failure
|
|
1822
|
+
return await self._emit_pending_retry_receipt(scenario, pending)
|
|
1823
|
+
next_index, next_runtime = next_leased
|
|
1824
|
+
return await self._run_scenario(
|
|
1825
|
+
scenario,
|
|
1826
|
+
scenario_index,
|
|
1827
|
+
attempt=2,
|
|
1828
|
+
tried=tried | {world_index},
|
|
1829
|
+
pre_leased=(next_index, next_runtime),
|
|
1830
|
+
pending_retry=pending,
|
|
1831
|
+
abort_holder=abort_holder,
|
|
1832
|
+
context=context,
|
|
1833
|
+
)
|
|
1834
|
+
|
|
1835
|
+
# M13: a plain terminal receipt is either a real passed/failed verdict (release — the
|
|
1836
|
+
# world is fine) or a non-retryable fault from `_fault()`. For the latter, an
|
|
1837
|
+
# exception/overrun code means the world is half-applied and must be discarded rather
|
|
1838
|
+
# than handed to the next scenario; `ready_not_ready` is a clean verdict and keeps
|
|
1839
|
+
# `release()`.
|
|
1840
|
+
if (
|
|
1841
|
+
outcome.failure is not None
|
|
1842
|
+
and outcome.failure.code in _DISCARD_ON_ERROR_CODES
|
|
1843
|
+
):
|
|
1844
|
+
await self._pool.mark_unhealthy(
|
|
1845
|
+
world_index, cause=outcome.failure.message
|
|
1846
|
+
)
|
|
1847
|
+
else:
|
|
1848
|
+
await self._pool.release(world_index)
|
|
1849
|
+
world_resolved = True
|
|
1850
|
+
await self._emit(self._outbound.receipt(outcome), what="receipt")
|
|
1851
|
+
return outcome
|
|
1852
|
+
finally:
|
|
1853
|
+
if not world_resolved:
|
|
1854
|
+
if self._pool.fenced is not None:
|
|
1855
|
+
# The exception that skipped every path above was a fence (or the pool was
|
|
1856
|
+
# already fenced by something else) -- the world itself never did anything
|
|
1857
|
+
# wrong. `mark_unhealthy()` here would emit a false `world_unhealthy` after the
|
|
1858
|
+
# run already stopped emitting, and schedule a `provision()` reconcile for a
|
|
1859
|
+
# job that is not coming back for it.
|
|
1860
|
+
await self._pool.release(world_index)
|
|
1861
|
+
else:
|
|
1862
|
+
# Something blew past every handled path above (a bug in this module
|
|
1863
|
+
# itself) — the world must not be silently stranded outside the pool's
|
|
1864
|
+
# bookkeeping. Discarded rather than released: an exception here leaves its
|
|
1865
|
+
# state unknown, and world-handle-interface.md's own exception rule is
|
|
1866
|
+
# "discarded and re-provisioned, never reused."
|
|
1867
|
+
await self._pool.mark_unhealthy(
|
|
1868
|
+
world_index,
|
|
1869
|
+
cause="scenario driver crashed while holding this world",
|
|
1870
|
+
)
|
|
1871
|
+
|
|
1872
|
+
async def _execute(
|
|
1873
|
+
self,
|
|
1874
|
+
scenario: Scenario,
|
|
1875
|
+
world: World,
|
|
1876
|
+
runtime: EnvironmentRuntime,
|
|
1877
|
+
world_index: int,
|
|
1878
|
+
*,
|
|
1879
|
+
attempt: int,
|
|
1880
|
+
context: "_ScenarioContext",
|
|
1881
|
+
) -> "ResultReceipt | _Retry":
|
|
1882
|
+
setup = await _run_phase(
|
|
1883
|
+
scenario.setup,
|
|
1884
|
+
world,
|
|
1885
|
+
timeout=SETUP_TIMEOUT_SECONDS,
|
|
1886
|
+
phase="setup",
|
|
1887
|
+
executor=self._executor,
|
|
1888
|
+
)
|
|
1889
|
+
if setup.failure is not None:
|
|
1890
|
+
return self._fault(
|
|
1891
|
+
scenario,
|
|
1892
|
+
world_index,
|
|
1893
|
+
attempt,
|
|
1894
|
+
setup.failure,
|
|
1895
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1896
|
+
)
|
|
1897
|
+
|
|
1898
|
+
read_only = world.read_only()
|
|
1899
|
+
ready = await _run_phase(
|
|
1900
|
+
scenario.ready,
|
|
1901
|
+
read_only,
|
|
1902
|
+
timeout=READY_TIMEOUT_SECONDS,
|
|
1903
|
+
phase="ready",
|
|
1904
|
+
executor=self._executor,
|
|
1905
|
+
)
|
|
1906
|
+
if ready.failure is not None:
|
|
1907
|
+
return self._fault(
|
|
1908
|
+
scenario,
|
|
1909
|
+
world_index,
|
|
1910
|
+
attempt,
|
|
1911
|
+
ready.failure,
|
|
1912
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1913
|
+
)
|
|
1914
|
+
verdict = _classify_ready(ready.value)
|
|
1915
|
+
if verdict.broken:
|
|
1916
|
+
return self._fault(
|
|
1917
|
+
scenario,
|
|
1918
|
+
world_index,
|
|
1919
|
+
attempt,
|
|
1920
|
+
_failure("ready_broken", f"ready() returned {ready.value!r}"),
|
|
1921
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1922
|
+
)
|
|
1923
|
+
if not verdict.held:
|
|
1924
|
+
# The verdict says the precondition did not hold, never whether the setup that was
|
|
1925
|
+
# supposed to establish it ran. Without that, a scenario that never dials looks the
|
|
1926
|
+
# same whether its setup failed or its check is wrong.
|
|
1927
|
+
logger.warning(
|
|
1928
|
+
"scenario %s not ready on world %s: %s (setup reported: %s)",
|
|
1929
|
+
scenario.scenario_key,
|
|
1930
|
+
world_index,
|
|
1931
|
+
verdict.reason or "no reason given",
|
|
1932
|
+
getattr(setup, "value", None),
|
|
1933
|
+
)
|
|
1934
|
+
return self._fault(
|
|
1935
|
+
scenario,
|
|
1936
|
+
world_index,
|
|
1937
|
+
attempt,
|
|
1938
|
+
_failure("ready_not_ready", verdict.reason or ""),
|
|
1939
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1940
|
+
)
|
|
1941
|
+
|
|
1942
|
+
try:
|
|
1943
|
+
call_outcome = await _run_call(self._call_runner, scenario, runtime, world)
|
|
1944
|
+
except WorldUnavailable as exc:
|
|
1945
|
+
return _Retry(
|
|
1946
|
+
_failure("world_unavailable", str(exc)),
|
|
1947
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1948
|
+
call=None,
|
|
1949
|
+
mark_unhealthy=True,
|
|
1950
|
+
)
|
|
1951
|
+
except CallAborted as exc:
|
|
1952
|
+
call = self._call_summary(exc.partial)
|
|
1953
|
+
return self._fault(
|
|
1954
|
+
scenario,
|
|
1955
|
+
world_index,
|
|
1956
|
+
attempt,
|
|
1957
|
+
_failure(exc.code, str(exc)),
|
|
1958
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1959
|
+
call=call,
|
|
1960
|
+
)
|
|
1961
|
+
except Exception as exc: # noqa: BLE001
|
|
1962
|
+
# B3: the call runner crashing outright (not a `CallAborted` it chose to raise) is the
|
|
1963
|
+
# same world-handle-interface.md v3.3 row — "the simulated-call machinery crashed" —
|
|
1964
|
+
# just with no partial evidence to report.
|
|
1965
|
+
return self._fault(
|
|
1966
|
+
scenario,
|
|
1967
|
+
world_index,
|
|
1968
|
+
attempt,
|
|
1969
|
+
_failure("call_failed", f"{type(exc).__name__}: {exc}"),
|
|
1970
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1971
|
+
call=None,
|
|
1972
|
+
)
|
|
1973
|
+
|
|
1974
|
+
# Set the moment the call step returns, so a crash later in this method (e.g.
|
|
1975
|
+
# `world.read_only()` below) still reports the call that genuinely ran, not `null`.
|
|
1976
|
+
context.call = self._call_summary(call_outcome)
|
|
1977
|
+
calls = list(
|
|
1978
|
+
call_outcome.calls
|
|
1979
|
+
) # m12: `folder.py::_RUNNABLE` expects a list, not a tuple.
|
|
1980
|
+
if not calls and getattr(scenario, "requires_tool_evidence", True):
|
|
1981
|
+
# M10: unconditioned on `turns` — an empty list must never reach checks regardless of
|
|
1982
|
+
# whether the simulator observed a turn (world-handle-interface.md "Coverage
|
|
1983
|
+
# guarantee": "An empty list is never handed to checks").
|
|
1984
|
+
failure = _failure(
|
|
1985
|
+
"evidence_missing",
|
|
1986
|
+
"no tool calls were captured for this scenario's call",
|
|
1987
|
+
)
|
|
1988
|
+
return _Retry(
|
|
1989
|
+
failure,
|
|
1990
|
+
sub_goals=_unjudged(scenario.sub_goals),
|
|
1991
|
+
call=self._call_summary(call_outcome),
|
|
1992
|
+
mark_unhealthy=False,
|
|
1993
|
+
)
|
|
1994
|
+
|
|
1995
|
+
if not scenario.sub_goals:
|
|
1996
|
+
# m7: `all(())` is vacuously True — a scenario declaring zero sub-goals must not read
|
|
1997
|
+
# as a silent pass.
|
|
1998
|
+
return self._fault(
|
|
1999
|
+
scenario,
|
|
2000
|
+
world_index,
|
|
2001
|
+
attempt,
|
|
2002
|
+
_failure(
|
|
2003
|
+
"check_broken",
|
|
2004
|
+
"scenario declared zero sub_goals — a vacuous pass is forbidden",
|
|
2005
|
+
),
|
|
2006
|
+
sub_goals=(),
|
|
2007
|
+
call=self._call_summary(call_outcome),
|
|
2008
|
+
)
|
|
2009
|
+
|
|
2010
|
+
sub_goal_results: list[SubGoalResult] = []
|
|
2011
|
+
judged_pending: list[tuple[int, Any]] = []
|
|
2012
|
+
check_handle = world.read_only()
|
|
2013
|
+
broken_failure: ReceiptFailure | None = None
|
|
2014
|
+
for goal in scenario.sub_goals:
|
|
2015
|
+
if broken_failure is not None:
|
|
2016
|
+
sub_goal_results.append(
|
|
2017
|
+
SubGoalResult(
|
|
2018
|
+
name=goal.name, held=None, reason=None, judged=goal.judged != ""
|
|
2019
|
+
)
|
|
2020
|
+
)
|
|
2021
|
+
continue
|
|
2022
|
+
outcome = await _run_phase(
|
|
2023
|
+
goal.check,
|
|
2024
|
+
check_handle,
|
|
2025
|
+
calls,
|
|
2026
|
+
timeout=CHECK_TIMEOUT_SECONDS,
|
|
2027
|
+
phase="check",
|
|
2028
|
+
executor=self._executor,
|
|
2029
|
+
)
|
|
2030
|
+
if outcome.failure is not None:
|
|
2031
|
+
if outcome.failure.code == "world_unavailable":
|
|
2032
|
+
return _Retry(
|
|
2033
|
+
outcome.failure,
|
|
2034
|
+
sub_goals=tuple(sub_goal_results)
|
|
2035
|
+
+ _unjudged([goal])
|
|
2036
|
+
+ _unjudged(scenario.sub_goals[len(sub_goal_results) + 1 :]),
|
|
2037
|
+
call=self._call_summary(call_outcome),
|
|
2038
|
+
mark_unhealthy=True,
|
|
2039
|
+
)
|
|
2040
|
+
broken_failure = outcome.failure
|
|
2041
|
+
sub_goal_results.append(
|
|
2042
|
+
SubGoalResult(
|
|
2043
|
+
name=goal.name, held=None, reason=None, judged=goal.judged != ""
|
|
2044
|
+
)
|
|
2045
|
+
)
|
|
2046
|
+
continue
|
|
2047
|
+
if goal.judged:
|
|
2048
|
+
# A judged sub-goal has no code to settle it: a model decides, here, while the
|
|
2049
|
+
# world the call left behind is still alive. Collected and run together below.
|
|
2050
|
+
judged_pending.append((len(sub_goal_results), goal))
|
|
2051
|
+
sub_goal_results.append(
|
|
2052
|
+
SubGoalResult(name=goal.name, held=None, reason=None, judged=True)
|
|
2053
|
+
)
|
|
2054
|
+
continue
|
|
2055
|
+
verdict = _classify_check(outcome.value)
|
|
2056
|
+
if verdict.broken:
|
|
2057
|
+
broken_failure = _failure(
|
|
2058
|
+
"check_broken", f"{goal.name}: check() returned {outcome.value!r}"
|
|
2059
|
+
)
|
|
2060
|
+
sub_goal_results.append(
|
|
2061
|
+
SubGoalResult(
|
|
2062
|
+
name=goal.name, held=None, reason=None, judged=goal.judged != ""
|
|
2063
|
+
)
|
|
2064
|
+
)
|
|
2065
|
+
continue
|
|
2066
|
+
sub_goal_results.append(
|
|
2067
|
+
SubGoalResult(
|
|
2068
|
+
name=goal.name,
|
|
2069
|
+
held=verdict.held,
|
|
2070
|
+
reason=_sub_goal_reason(goal, verdict),
|
|
2071
|
+
judged=goal.judged != "",
|
|
2072
|
+
)
|
|
2073
|
+
)
|
|
2074
|
+
|
|
2075
|
+
if judged_pending:
|
|
2076
|
+
# Judged sub-goals only read, so they are independent of each other and of the coded
|
|
2077
|
+
# checks: one round trip for all of them rather than one each.
|
|
2078
|
+
async def _settle(goal: Any) -> Any:
|
|
2079
|
+
# Awaited, not called inline: calling an injected judge whose signature does not
|
|
2080
|
+
# match raises while the coroutines are still being built, which is outside
|
|
2081
|
+
# `gather`'s net and errors the scenario. Inside a coroutine it is just a fault.
|
|
2082
|
+
return await self._judge(
|
|
2083
|
+
goal, check_handle, calls, messages=call_outcome.messages
|
|
2084
|
+
)
|
|
2085
|
+
|
|
2086
|
+
verdicts = await asyncio.gather(
|
|
2087
|
+
*(_settle(goal) for _, goal in judged_pending),
|
|
2088
|
+
return_exceptions=True,
|
|
2089
|
+
)
|
|
2090
|
+
for (slot, goal), outcome in zip(judged_pending, verdicts):
|
|
2091
|
+
if isinstance(outcome, BaseException):
|
|
2092
|
+
held, why = None, f"the judge could not run: {outcome!r}"
|
|
2093
|
+
else:
|
|
2094
|
+
try:
|
|
2095
|
+
held, why = outcome
|
|
2096
|
+
except (TypeError, ValueError):
|
|
2097
|
+
# An injected judge that answers in some other shape is unreadable, not
|
|
2098
|
+
# authoritative. Unpacking it here would raise inside `_grade` and error
|
|
2099
|
+
# the whole scenario, which is the one thing a verdict must never do.
|
|
2100
|
+
held, why = (
|
|
2101
|
+
None,
|
|
2102
|
+
f"the judge returned no usable verdict: {outcome!r}",
|
|
2103
|
+
)
|
|
2104
|
+
sub_goal_results[slot] = SubGoalResult(
|
|
2105
|
+
name=goal.name, held=held, reason=why, judged=True
|
|
2106
|
+
)
|
|
2107
|
+
|
|
2108
|
+
if broken_failure is not None:
|
|
2109
|
+
return self._fault(
|
|
2110
|
+
scenario,
|
|
2111
|
+
world_index,
|
|
2112
|
+
attempt,
|
|
2113
|
+
broken_failure,
|
|
2114
|
+
sub_goals=tuple(sub_goal_results),
|
|
2115
|
+
call=self._call_summary(call_outcome),
|
|
2116
|
+
)
|
|
2117
|
+
|
|
2118
|
+
# A sub-goal the judge did not settle is reported unsettled on the sub-goal itself and
|
|
2119
|
+
# never decides the scenario: the call ran, its evidence stands, and a model that could
|
|
2120
|
+
# not answer is a fault of neither the agent nor the run. Only a settled `False` fails a
|
|
2121
|
+
# scenario. `errored` stays reachable for a call or infrastructure fault, which is raised
|
|
2122
|
+
# elsewhere; nothing about a verdict produces one.
|
|
2123
|
+
if any(result.held is False for result in sub_goal_results):
|
|
2124
|
+
status = "failed"
|
|
2125
|
+
else:
|
|
2126
|
+
status = "passed"
|
|
2127
|
+
failure = None
|
|
2128
|
+
return ResultReceipt(
|
|
2129
|
+
scenario_key=scenario.scenario_key,
|
|
2130
|
+
scenario_id=scenario.scenario_id,
|
|
2131
|
+
scenario_attempt=attempt,
|
|
2132
|
+
world_index=world_index,
|
|
2133
|
+
status=status,
|
|
2134
|
+
sub_goals=tuple(sub_goal_results),
|
|
2135
|
+
evaluations=(),
|
|
2136
|
+
call=self._call_summary(call_outcome),
|
|
2137
|
+
failure=failure,
|
|
2138
|
+
)
|
|
2139
|
+
|
|
2140
|
+
@staticmethod
|
|
2141
|
+
def _call_summary(outcome: CallOutcome | None) -> CallSummary | None:
|
|
2142
|
+
if outcome is None:
|
|
2143
|
+
return None
|
|
2144
|
+
return CallSummary(
|
|
2145
|
+
started_at=outcome.started_at,
|
|
2146
|
+
ended_at=outcome.ended_at,
|
|
2147
|
+
duration_ms=outcome.duration_ms,
|
|
2148
|
+
turns=outcome.turns,
|
|
2149
|
+
transcript_artifact=outcome.transcript_artifact,
|
|
2150
|
+
recording_artifacts=outcome.recording_artifacts,
|
|
2151
|
+
stop_reason=outcome.stop_reason,
|
|
2152
|
+
)
|
|
2153
|
+
|
|
2154
|
+
def _fault(
|
|
2155
|
+
self,
|
|
2156
|
+
scenario: Scenario,
|
|
2157
|
+
world_index: int,
|
|
2158
|
+
attempt: int,
|
|
2159
|
+
failure: ReceiptFailure,
|
|
2160
|
+
*,
|
|
2161
|
+
sub_goals: tuple[SubGoalResult, ...],
|
|
2162
|
+
call: CallSummary | None = None,
|
|
2163
|
+
retry: bool | None = None,
|
|
2164
|
+
) -> "ResultReceipt | _Retry":
|
|
2165
|
+
should_retry = _is_retryable(failure.code) if retry is None else retry
|
|
2166
|
+
if should_retry:
|
|
2167
|
+
return _Retry(
|
|
2168
|
+
failure,
|
|
2169
|
+
sub_goals=sub_goals,
|
|
2170
|
+
call=call,
|
|
2171
|
+
mark_unhealthy=failure.code == "world_unavailable",
|
|
2172
|
+
)
|
|
2173
|
+
return ResultReceipt(
|
|
2174
|
+
scenario_key=scenario.scenario_key,
|
|
2175
|
+
scenario_id=scenario.scenario_id,
|
|
2176
|
+
scenario_attempt=attempt,
|
|
2177
|
+
world_index=world_index,
|
|
2178
|
+
status="errored",
|
|
2179
|
+
sub_goals=sub_goals,
|
|
2180
|
+
evaluations=(),
|
|
2181
|
+
call=call,
|
|
2182
|
+
failure=failure,
|
|
2183
|
+
)
|
|
2184
|
+
|
|
2185
|
+
|
|
2186
|
+
@dataclass(frozen=True)
|
|
2187
|
+
class _Retry:
|
|
2188
|
+
failure: ReceiptFailure
|
|
2189
|
+
sub_goals: tuple[SubGoalResult, ...]
|
|
2190
|
+
call: CallSummary | None
|
|
2191
|
+
mark_unhealthy: bool
|
|
2192
|
+
|
|
2193
|
+
|
|
2194
|
+
__all__ = [
|
|
2195
|
+
"CHECK_TIMEOUT_SECONDS",
|
|
2196
|
+
"READY_TIMEOUT_SECONDS",
|
|
2197
|
+
"SETUP_TIMEOUT_SECONDS",
|
|
2198
|
+
"Call",
|
|
2199
|
+
"CallAborted",
|
|
2200
|
+
"CallOutcome",
|
|
2201
|
+
"CallRunner",
|
|
2202
|
+
"CallSummary",
|
|
2203
|
+
"Evaluation",
|
|
2204
|
+
"HostedScheduler",
|
|
2205
|
+
"NoWorldsAvailable",
|
|
2206
|
+
"OutboundPort",
|
|
2207
|
+
"ReadOnlyWorld",
|
|
2208
|
+
"ReceiptFailure",
|
|
2209
|
+
"ResultReceipt",
|
|
2210
|
+
"RunResult",
|
|
2211
|
+
"Scenario",
|
|
2212
|
+
"SubGoal",
|
|
2213
|
+
"SubGoalResult",
|
|
2214
|
+
"World",
|
|
2215
|
+
"WorldFactory",
|
|
2216
|
+
"WorldPool",
|
|
2217
|
+
"WorldProvisioner",
|
|
2218
|
+
]
|