agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,879 @@
|
|
|
1
|
+
"""The scenario-source adapter: reads generated scenario documents out of the bundle (code-as-text,
|
|
2
|
+
on the on-disk layout `folder.py` documents) and turns them into the `Scenario`/`SubGoal` objects
|
|
3
|
+
`hosted_scheduler.py` actually drives.
|
|
4
|
+
|
|
5
|
+
Deliberately does NOT import `fi.alk.harness.folder` or `fi.alk.harness.scenario` for the model:
|
|
6
|
+
both exist at HEAD, but HEAD's `Scenario` carries no `scenario_key`/`scenario_id` (those are
|
|
7
|
+
pr63-only) and its default `extra="ignore"` would silently discard exactly the two fields the
|
|
8
|
+
scheduler needs off a `scenario.json` written in the newer shape. So this module reads
|
|
9
|
+
`scenario.json` as a plain dict and pulls fields out by key, mirroring the documented layout
|
|
10
|
+
instead of depending on either model -- see the report's design-decisions section for the
|
|
11
|
+
consequences of that choice (HEAD-model drift).
|
|
12
|
+
|
|
13
|
+
RESOLVED (p13-worker-r2, reports/p13-worker-r2.md CONTRACT NOTES): the `provision`/`begin` wire
|
|
14
|
+
shapes below follow the platform's actual, live route (futureagi/simulate/serializers/services/
|
|
15
|
+
views `hosted_harness.py`) rather than the Scenario Generation Contract text (PR #63), where
|
|
16
|
+
the two disagree -- a single `POST .../scenarios/` discriminated by a body-level `operation` field,
|
|
17
|
+
`begin` keyed on the full `scenario_keys` set, and a provision response KEYED by `scenario_key`
|
|
18
|
+
(never a position-ordered array). `register_with_platform` below is the seam that builds those
|
|
19
|
+
payloads and merges the platform-assigned `scenario_id`s back onto each scenario.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import asyncio
|
|
25
|
+
import json
|
|
26
|
+
import logging
|
|
27
|
+
from dataclasses import dataclass, field, replace
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
from typing import TYPE_CHECKING, Any, Callable, Sequence
|
|
30
|
+
|
|
31
|
+
from . import outbound as ob
|
|
32
|
+
from .catalogue import CATALOGUE
|
|
33
|
+
from .job import FailureDomain
|
|
34
|
+
|
|
35
|
+
if TYPE_CHECKING:
|
|
36
|
+
from .hosted_entrypoint import ScenariosClient
|
|
37
|
+
|
|
38
|
+
# LAYOUT DECISION (contract-silent -- hosted-execution-seams.md v1.15 §2 never mentions scenario
|
|
39
|
+
# documents, and §7 assigns the on-disk layout to that contract, status "in review"). Scenario
|
|
40
|
+
# documents live at `<bundle_dir>/<SCENARIOS_DIRNAME>/<name>/...`, matching `folder.py`'s own
|
|
41
|
+
# `SCENARIOS` constant, so a write_folder destination of `<bundle_dir>` lands correctly with no
|
|
42
|
+
# translation. Kept as one module-level constant so a later contract can move it in one edit.
|
|
43
|
+
logger = logging.getLogger(__name__)
|
|
44
|
+
|
|
45
|
+
SCENARIOS_DIRNAME = "scenarios"
|
|
46
|
+
|
|
47
|
+
_CHECKS_DIRNAME = "checks"
|
|
48
|
+
_SCENARIO_JSON = "scenario.json"
|
|
49
|
+
_CATALOGUE_JSON = "sub_goals.json"
|
|
50
|
+
_SETUP_PY = "setup.py"
|
|
51
|
+
_READY_PY = "ready.py"
|
|
52
|
+
|
|
53
|
+
# R1-5: divergence (a) (compile once, at load) moves every scenario file's compile+exec into
|
|
54
|
+
# `asyncio.to_thread(load_scenarios, ...)`, OUTSIDE every phase budget the scheduler enforces
|
|
55
|
+
# (`SETUP_TIMEOUT_SECONDS`/`READY_TIMEOUT_SECONDS`/`CHECK_TIMEOUT_SECONDS` all apply downstream, to
|
|
56
|
+
# `_run_phase`). Pathological module-level code (`while True: pass`, a blocking socket read) would
|
|
57
|
+
# otherwise hang the load with zero terminal events, no timeout catching it, ever. One generous
|
|
58
|
+
# wall-clock budget here converts that hang into a typed terminal instead -- a worker thread cannot
|
|
59
|
+
# actually be killed, so this accepts a leaked thread over an unbounded one, on the reasoning that a
|
|
60
|
+
# terminal event today is strictly better than none ever.
|
|
61
|
+
_LOAD_TIMEOUT_SECONDS = 60.0
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# Every scenario a job asked for is called. A job that wrote two hundred is asking for two hundred
|
|
65
|
+
# calls: that is the product, and calling fewer would quietly deliver a fraction of what somebody
|
|
66
|
+
# paid for. There is deliberately no setting here; a smaller run is a smaller `scenario_count`.
|
|
67
|
+
def sampled_for_calling(scenarios: Sequence[Any]) -> list[Any]:
|
|
68
|
+
"""Which scenarios this job calls, which is all of them."""
|
|
69
|
+
return list(scenarios)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# The TEXT of a judged sub-goal's check is never persisted by `folder.py`'s `write_folder` (only
|
|
73
|
+
# `SubGoal.deterministic()` entries get a `checks/<name>.py` file) -- this fixed marker stands in
|
|
74
|
+
# for it so `SubGoal.judged` (mandatory; read by plain attribute access in `hosted_scheduler.py`)
|
|
75
|
+
# is never empty for a sub-goal this reader knows is judged. The round-trip loss this represents is
|
|
76
|
+
# recorded under CONTRACT QUESTIONS in the report.
|
|
77
|
+
_JUDGED_MARKER = "judged (reason not persisted by folder.py's on-disk layout)"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class ScenarioDocumentInvalid(RuntimeError):
|
|
81
|
+
"""A scenario folder under `<bundle_dir>/scenarios/` is unreadable or malformed: missing
|
|
82
|
+
`scenario.json`, invalid JSON, a field of the wrong shape, or a `setup.py`/`ready.py`/
|
|
83
|
+
`checks/<goal>.py` that will not compile. Raised rather than skipped -- `folder.py`'s own
|
|
84
|
+
`read_all` swallows exactly this and continues, which is right for a human editing a suite by
|
|
85
|
+
hand and wrong for a hosted job, where a bad scenario silently vanishing from the run reads as
|
|
86
|
+
a suite that passed with fewer scenarios than it should have."""
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def bundle_has_scenarios(bundle_dir: Path) -> bool:
|
|
90
|
+
"""The LAYOUT DECISION's presence test: `<bundle_dir>/scenarios/` exists and at least one of
|
|
91
|
+
its subdirectories holds a `scenario.json`. Deliberately narrow -- an empty or missing
|
|
92
|
+
`scenarios/` must not flip the wiring decision away from the safe `NotWiredScenarioSource`
|
|
93
|
+
default. A bundle that means to carry scenarios but got the layout wrong fails loudly once
|
|
94
|
+
`load_scenarios` actually reads it, not silently by being treated as scenario-free here.
|
|
95
|
+
|
|
96
|
+
An unreadable `scenarios/` directory (permission denied, race with deletion, etc.) reports
|
|
97
|
+
False rather than raising `OSError` (R1-1): this call sits in `hosted_entrypoint.py`'s wiring
|
|
98
|
+
`if` BEFORE the `try`/`except` that maps `ScenarioDocumentInvalid` to a typed terminal, so an
|
|
99
|
+
escape here would kill the whole guest process with no terminal event at all. Falling back to
|
|
100
|
+
`NotWiredScenarioSource`'s existing typed failure is the safe direction -- the alternative of
|
|
101
|
+
raising here has nowhere typed to land.
|
|
102
|
+
"""
|
|
103
|
+
root = bundle_dir / SCENARIOS_DIRNAME
|
|
104
|
+
if not root.is_dir():
|
|
105
|
+
return False
|
|
106
|
+
try:
|
|
107
|
+
children = list(root.iterdir())
|
|
108
|
+
except OSError:
|
|
109
|
+
return False
|
|
110
|
+
return any(
|
|
111
|
+
(child / _SCENARIO_JSON).is_file() for child in children if child.is_dir()
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _judged_placeholder_check(world: Any, calls: Any) -> None:
|
|
116
|
+
"""The check for a judged (non-deterministic) sub-goal: always reports "held", via the same
|
|
117
|
+
convention a deterministic check uses. `SubGoalResult.judged` -- not this return value -- is
|
|
118
|
+
what has to tell downstream a real judge still has to run; see CONTRACT QUESTIONS in the
|
|
119
|
+
report for the gap that leaves (a judged-only scenario reports "passed" before any judge runs).
|
|
120
|
+
"""
|
|
121
|
+
del world, calls
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _compile_entry(
|
|
126
|
+
source: str, *, label: str, entry: str, allow_empty: bool = True
|
|
127
|
+
) -> Callable[..., object]:
|
|
128
|
+
"""One scenario code-text -> a bare callable that raises, compiled ONCE here rather than per
|
|
129
|
+
call. Mirrors `folder.py`'s `_run` in exactly two respects: `compile(source, name, "exec")`
|
|
130
|
+
into a fresh, empty-dict namespace with default builtins, and (when `allow_empty`)
|
|
131
|
+
empty/whitespace-only source is a no-op success. Deliberately diverges from `_run` in the two
|
|
132
|
+
respects the brief calls out:
|
|
133
|
+
(a) compiling here, at load, turns a syntax error into one typed terminal for the whole job
|
|
134
|
+
instead of a per-scenario fault discovered mid-run; (b) the compiled function is returned
|
|
135
|
+
BARE, never wrapped in `_run`'s `Outcome` -- `hosted_scheduler.py`'s `_run_phase` is what
|
|
136
|
+
classifies a raised exception into `setup_crashed`/`ready_broken`/`check_broken`/a timeout, and
|
|
137
|
+
an `Outcome` return here would swallow every one of those before `_run_phase` ever saw it.
|
|
138
|
+
`_run`'s complaint-sentence return convention (a non-None, non-True, non-empty-string value
|
|
139
|
+
means "did not hold") is left untranslated for the same reason: `hosted_scheduler.py`'s own
|
|
140
|
+
`_classify_ready`/`_classify_check` already own that classification on the return-VALUE side of
|
|
141
|
+
this boundary; only the raise-vs-return boundary belongs to this module.
|
|
142
|
+
|
|
143
|
+
`allow_empty=False` is for `check` entries only (R1-2): an EXISTING `checks/<name>.py` that is
|
|
144
|
+
empty or whitespace-only compiles to nothing, and handing back a no-op "held" callable -- the
|
|
145
|
+
right behavior for "no setup/ready code here" -- would silently turn "there is no check" into
|
|
146
|
+
"the check passed": a vacuous deterministic pass, which is exactly what `hosted_scheduler.py`
|
|
147
|
+
forbids one scenario level up ("scenario declared zero sub_goals"). Absence of the file
|
|
148
|
+
entirely is what means "judged" (see `_load_one`); an existing-but-empty file is malformed.
|
|
149
|
+
"""
|
|
150
|
+
if not source.strip():
|
|
151
|
+
if allow_empty:
|
|
152
|
+
return lambda *args: None
|
|
153
|
+
raise ScenarioDocumentInvalid(f"{label} defines no {entry}()")
|
|
154
|
+
try:
|
|
155
|
+
code = compile(source, f"<{label}>", "exec")
|
|
156
|
+
except (SyntaxError, ValueError) as exc:
|
|
157
|
+
# SyntaxError is the common case; a NUL byte in the source raises ValueError on some
|
|
158
|
+
# interpreter versions (R1-1) rather than SyntaxError -- both are the same content defect.
|
|
159
|
+
raise ScenarioDocumentInvalid(f"{label} would not compile: {exc}") from exc
|
|
160
|
+
namespace: dict[str, Any] = {}
|
|
161
|
+
try:
|
|
162
|
+
exec(code, namespace) # noqa: S102 - scenario code is meant to be exec'd; see CONTRACT QUESTIONS
|
|
163
|
+
except (Exception, SystemExit, KeyboardInterrupt) as exc: # noqa: BLE001 - see R1-1
|
|
164
|
+
# A module-level `sys.exit()`/`raise SystemExit(...)` in the file itself is a malformed
|
|
165
|
+
# document, not a request to shut the guest process down -- `SystemExit`/`KeyboardInterrupt`
|
|
166
|
+
# are `BaseException`, not `Exception`, so a bare `except Exception` (the pre-R1-1 shape)
|
|
167
|
+
# let them straight through this boundary and out of `run_job` with zero terminal events,
|
|
168
|
+
# the guest exiting with whatever code the scenario file itself chose.
|
|
169
|
+
raise ScenarioDocumentInvalid(f"{label} would not compile: {exc}") from exc
|
|
170
|
+
function = namespace.get(entry)
|
|
171
|
+
if not callable(function):
|
|
172
|
+
raise ScenarioDocumentInvalid(f"{label} defines no {entry}()")
|
|
173
|
+
return function
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@dataclass(frozen=True)
|
|
177
|
+
class _CompiledSubGoal:
|
|
178
|
+
"""Satisfies `hosted_scheduler.SubGoal`: `name`/`judged` as plain attributes, `check` as a bare
|
|
179
|
+
callable. Stored as instance DATA rather than a `def check(self, world, calls)` method so
|
|
180
|
+
`goal.check(world, calls)` invokes the compiled function directly with exactly the two
|
|
181
|
+
positional arguments `_run_phase` passes -- a real method would prepend `self` as a third."""
|
|
182
|
+
|
|
183
|
+
name: str
|
|
184
|
+
judged: str
|
|
185
|
+
check: Callable[[Any, Any], object]
|
|
186
|
+
what: str = ""
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
@dataclass(frozen=True)
|
|
190
|
+
class _CompiledScenario:
|
|
191
|
+
"""Satisfies `hosted_scheduler.Scenario`. `scenario_key`/`scenario_id` are carried VERBATIM
|
|
192
|
+
from the document, including an empty `scenario_id` -- synthesizing one here would hide that
|
|
193
|
+
pre-allocation has not actually run (see CONTRACT QUESTIONS: receipts carry `scenario_id ""`
|
|
194
|
+
until that seam is wired). `setup`/`ready` are likewise stored as data, for the same reason as
|
|
195
|
+
`_CompiledSubGoal.check` above."""
|
|
196
|
+
|
|
197
|
+
scenario_key: str
|
|
198
|
+
scenario_id: str
|
|
199
|
+
sub_goals: tuple[_CompiledSubGoal, ...]
|
|
200
|
+
setup: Callable[[Any], object]
|
|
201
|
+
ready: Callable[[Any], object]
|
|
202
|
+
requires_tool_evidence: bool = True
|
|
203
|
+
# Who this person is and what they came for, as the platform's own persona record. Read off the
|
|
204
|
+
# same document and sent at pre-allocation, so a call can be read on the platform without the
|
|
205
|
+
# scenario file beside it. Presentation only: nothing in the scheduler looks at it.
|
|
206
|
+
presented: dict[str, Any] = field(default_factory=dict)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _read_text(path: Path, *, label: str) -> str:
|
|
210
|
+
"""Missing is "" (mirrors `folder.py`'s own missing-setup/ready-is-empty convention); present
|
|
211
|
+
but unreadable (permission denied, a directory instead of a file) or present but not valid
|
|
212
|
+
UTF-8 is a typed `ScenarioDocumentInvalid`, never a raw `OSError`/`UnicodeDecodeError` escaping
|
|
213
|
+
this module (R1-1) -- both are equally "this scenario folder is malformed", the same
|
|
214
|
+
conclusion `_load_one`'s other reads already reach for a bad `scenario.json`.
|
|
215
|
+
"""
|
|
216
|
+
if not path.exists():
|
|
217
|
+
return ""
|
|
218
|
+
try:
|
|
219
|
+
return path.read_text(encoding="utf-8")
|
|
220
|
+
except (OSError, UnicodeDecodeError) as exc:
|
|
221
|
+
raise ScenarioDocumentInvalid(
|
|
222
|
+
f"{label}: cannot read {path.name}: {exc}"
|
|
223
|
+
) from exc
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _validate_subgoal_name(name: str, *, folder_name: str) -> None:
|
|
227
|
+
"""R1-3: `sub_goals[]` entries are used verbatim to build `checks/<name>.py` -- a path
|
|
228
|
+
separator or a `..` segment lets a name escape `checks/` (and the sealed bundle) entirely: an
|
|
229
|
+
absolute name execs an arbitrary file never hashed into the bundle's `files[]` (bypassing the
|
|
230
|
+
§2e integrity seal), and a `../`-style traversal that resolves to nothing silently turns into a
|
|
231
|
+
JUDGED sub-goal (`check_path.is_file()` is False) instead of a typed failure. Rejecting
|
|
232
|
+
anything but a plain filename component closes both."""
|
|
233
|
+
if not name or "/" in name or "\\" in name or name in (".", ".."):
|
|
234
|
+
raise ScenarioDocumentInvalid(
|
|
235
|
+
f"{folder_name}: sub_goals name {name!r} is not a plain filename "
|
|
236
|
+
"(no path separators, no '..', no leading '/')"
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _presented(body: dict[str, Any], *, scenario_key: str) -> dict[str, Any]:
|
|
241
|
+
"""The persona record the platform stores for a scenario, built by the platform module's own
|
|
242
|
+
helper rather than a second copy of it here, so both paths describe a person the same way.
|
|
243
|
+
|
|
244
|
+
Presentation only, and never load-bearing: anything missing from the document is simply absent
|
|
245
|
+
from what the platform shows, and a document this cannot read still runs.
|
|
246
|
+
"""
|
|
247
|
+
from types import SimpleNamespace
|
|
248
|
+
|
|
249
|
+
from .platform import persona_of
|
|
250
|
+
|
|
251
|
+
# Derive a short display name for the scenario. The document may carry
|
|
252
|
+
# an explicit ``use_case``; when it does not, the scenario_key slug
|
|
253
|
+
# (e.g. ``refuse-booking-suspended-account``) is humanised so
|
|
254
|
+
# ``display_scenario_name`` never falls through to the long instruction
|
|
255
|
+
# text that ``name`` often contains in voice scenarios.
|
|
256
|
+
use_case = str(body.get("use_case") or "").strip()
|
|
257
|
+
if not use_case and scenario_key:
|
|
258
|
+
use_case = scenario_key.replace("-", " ").replace("_", " ")
|
|
259
|
+
use_case = use_case[:1].upper() + use_case[1:]
|
|
260
|
+
|
|
261
|
+
try:
|
|
262
|
+
return persona_of(
|
|
263
|
+
SimpleNamespace(
|
|
264
|
+
persona=body.get("persona") or {},
|
|
265
|
+
name=str(body.get("name") or ""),
|
|
266
|
+
scenario_key=scenario_key,
|
|
267
|
+
instruction=str(body.get("instruction") or ""),
|
|
268
|
+
tests=str(body.get("tests") or ""),
|
|
269
|
+
use_case=use_case,
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
except Exception: # noqa: BLE001 - a scenario never fails to run over how it is displayed
|
|
273
|
+
return {}
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _deterministic_names(bundle_dir: Path) -> set[str]:
|
|
277
|
+
"""Sub-goals the catalogue settles in code, so a missing check file is a defect and not a judge.
|
|
278
|
+
|
|
279
|
+
Absence of `checks/<name>.py` is what marks a sub-goal judged, which is right when the catalogue
|
|
280
|
+
says nobody can settle it in code and wrong when the check simply never reached the folder: the
|
|
281
|
+
run then reports a judged verdict for something that was meant to be measured, and a judge asked
|
|
282
|
+
about state it cannot see tends to say yes.
|
|
283
|
+
"""
|
|
284
|
+
path = bundle_dir / CATALOGUE
|
|
285
|
+
if not path.is_file():
|
|
286
|
+
return set()
|
|
287
|
+
try:
|
|
288
|
+
body = json.loads(path.read_text(encoding="utf-8"))
|
|
289
|
+
return {
|
|
290
|
+
str(one.get("name") or "")
|
|
291
|
+
for one in (body.get("sub_goals") or [])
|
|
292
|
+
if str(one.get("check") or "").strip()
|
|
293
|
+
} - {""}
|
|
294
|
+
except Exception: # noqa: BLE001 - an unreadable catalogue leaves the old behaviour
|
|
295
|
+
return set()
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _declared_tool_names(bundle_dir: Path) -> set[str]:
|
|
299
|
+
"""Return the target tools declared by the authored contract.
|
|
300
|
+
|
|
301
|
+
The solution format intentionally permits semantic steps that are not executable tools. The
|
|
302
|
+
contract's tool inventory is therefore the only stable way to decide whether a runtime tool
|
|
303
|
+
trace is required. A malformed or absent inventory is treated as empty here; contract
|
|
304
|
+
validation owns reporting malformed contract content, while this reader must not invent tool
|
|
305
|
+
requirements that the target itself never declared.
|
|
306
|
+
"""
|
|
307
|
+
path = bundle_dir / "contract.json"
|
|
308
|
+
if not path.is_file():
|
|
309
|
+
return set()
|
|
310
|
+
try:
|
|
311
|
+
body = json.loads(path.read_text(encoding="utf-8"))
|
|
312
|
+
tools = body.get("tools") if isinstance(body, dict) else None
|
|
313
|
+
if not isinstance(tools, list):
|
|
314
|
+
return set()
|
|
315
|
+
return {
|
|
316
|
+
str(tool.get("name") or "").strip()
|
|
317
|
+
for tool in tools
|
|
318
|
+
if isinstance(tool, dict)
|
|
319
|
+
} - {""}
|
|
320
|
+
except Exception: # noqa: BLE001 - contract validation reports the content defect
|
|
321
|
+
return set()
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _load_catalogue_claims(bundle_dir: Path) -> dict[str, dict[str, str]]:
|
|
325
|
+
"""`sub_goals.json`'s `what`/`judged` text, which `folder.py` never writes into a scenario
|
|
326
|
+
folder. Without it a judged sub-goal reaches the platform as a name and nothing to decide.
|
|
327
|
+
"""
|
|
328
|
+
path = bundle_dir / _CATALOGUE_JSON
|
|
329
|
+
if not path.is_file():
|
|
330
|
+
# Without this every sub-goal reaches the platform with no description, so a pass explains
|
|
331
|
+
# itself as "the check found nothing wrong" and a judged one arrives with nothing to
|
|
332
|
+
# decide. Said out loud because the symptom shows up two systems away from the cause.
|
|
333
|
+
logger.warning(
|
|
334
|
+
"no %s beside the scenarios in %s: sub-goals will carry no description or claim",
|
|
335
|
+
_CATALOGUE_JSON,
|
|
336
|
+
bundle_dir,
|
|
337
|
+
)
|
|
338
|
+
return {}
|
|
339
|
+
try:
|
|
340
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
341
|
+
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
|
|
342
|
+
return {}
|
|
343
|
+
entries = raw.get("sub_goals") if isinstance(raw, dict) else None
|
|
344
|
+
if not isinstance(entries, list):
|
|
345
|
+
return {}
|
|
346
|
+
claims: dict[str, dict[str, str]] = {}
|
|
347
|
+
for entry in entries:
|
|
348
|
+
if not isinstance(entry, dict) or not isinstance(entry.get("name"), str):
|
|
349
|
+
continue
|
|
350
|
+
claims[entry["name"]] = {
|
|
351
|
+
"what": str(entry.get("what") or ""),
|
|
352
|
+
"judged": str(entry.get("judged") or ""),
|
|
353
|
+
}
|
|
354
|
+
return claims
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _load_one(
|
|
358
|
+
folder: Path,
|
|
359
|
+
*,
|
|
360
|
+
settled_in_code: set[str] | None = None,
|
|
361
|
+
declared_tools: set[str] | None = None,
|
|
362
|
+
) -> _CompiledScenario:
|
|
363
|
+
"""One scenario folder -> a `Scenario`-protocol object. Mirrors `folder.py`'s documented
|
|
364
|
+
layout (`scenario.json` + `setup.py` + `ready.py` + `checks/<goal>.py`) but reads
|
|
365
|
+
`scenario.json` itself as a plain dict rather than through `fi.alk.harness.scenario.Scenario`
|
|
366
|
+
-- see the module docstring. `folder.py`'s `read_folder` restores only `setup_code`/
|
|
367
|
+
`ready_code` from a folder; it does not read `checks/` at all, so every `checks/<goal>.py` for
|
|
368
|
+
each name in the document's `sub_goals` is read here, by this module, directly.
|
|
369
|
+
"""
|
|
370
|
+
body_path = folder / _SCENARIO_JSON
|
|
371
|
+
try:
|
|
372
|
+
raw = body_path.read_text(encoding="utf-8")
|
|
373
|
+
except (OSError, UnicodeDecodeError) as exc:
|
|
374
|
+
# UnicodeDecodeError is a ValueError, not an OSError -- widened alongside it (R1-1) so a
|
|
375
|
+
# non-UTF-8 `scenario.json` is the same typed failure as an unreadable one, not an escape.
|
|
376
|
+
raise ScenarioDocumentInvalid(
|
|
377
|
+
f"{folder.name}: cannot read {_SCENARIO_JSON}: {exc}"
|
|
378
|
+
) from exc
|
|
379
|
+
try:
|
|
380
|
+
body = json.loads(raw)
|
|
381
|
+
except json.JSONDecodeError as exc:
|
|
382
|
+
raise ScenarioDocumentInvalid(
|
|
383
|
+
f"{folder.name}: {_SCENARIO_JSON} is not valid JSON: {exc}"
|
|
384
|
+
) from exc
|
|
385
|
+
if not isinstance(body, dict):
|
|
386
|
+
raise ScenarioDocumentInvalid(
|
|
387
|
+
f"{folder.name}: {_SCENARIO_JSON} is not a JSON object"
|
|
388
|
+
)
|
|
389
|
+
|
|
390
|
+
scenario_key = body.get("scenario_key", "")
|
|
391
|
+
if not isinstance(scenario_key, str):
|
|
392
|
+
raise ScenarioDocumentInvalid(f"{folder.name}: scenario_key is not a string")
|
|
393
|
+
scenario_id = body.get("scenario_id", "")
|
|
394
|
+
if not isinstance(scenario_id, str):
|
|
395
|
+
raise ScenarioDocumentInvalid(f"{folder.name}: scenario_id is not a string")
|
|
396
|
+
sub_goal_names = body.get("sub_goals", [])
|
|
397
|
+
if not isinstance(sub_goal_names, list) or not all(
|
|
398
|
+
isinstance(name, str) for name in sub_goal_names
|
|
399
|
+
):
|
|
400
|
+
raise ScenarioDocumentInvalid(
|
|
401
|
+
f"{folder.name}: sub_goals is not a list of strings"
|
|
402
|
+
)
|
|
403
|
+
for name in sub_goal_names:
|
|
404
|
+
_validate_subgoal_name(name, folder_name=folder.name)
|
|
405
|
+
|
|
406
|
+
solution = body.get("solution", [])
|
|
407
|
+
if not isinstance(solution, list) or not all(
|
|
408
|
+
isinstance(step, dict) for step in solution
|
|
409
|
+
):
|
|
410
|
+
raise ScenarioDocumentInvalid(
|
|
411
|
+
f"{folder.name}: solution is not a list of objects"
|
|
412
|
+
)
|
|
413
|
+
solution_tools = [step.get("tool") for step in solution]
|
|
414
|
+
if not all(isinstance(tool, str) and tool.strip() for tool in solution_tools):
|
|
415
|
+
raise ScenarioDocumentInvalid(
|
|
416
|
+
f"{folder.name}: every solution step must name a non-empty tool"
|
|
417
|
+
)
|
|
418
|
+
# A solution may contain semantic narration steps (for example ``listen`` and ``respond``)
|
|
419
|
+
# alongside real agent tool calls. The contract is the authoritative inventory of tools the
|
|
420
|
+
# target can actually emit, so evidence is required exactly when a solution references one of
|
|
421
|
+
# those declared tools. Inferring this from arbitrary step names creates false infrastructure
|
|
422
|
+
# failures for conversational agents and requires an ever-growing pseudo-tool denylist.
|
|
423
|
+
requires_tool_evidence = bool(set(solution_tools) & (declared_tools or set()))
|
|
424
|
+
|
|
425
|
+
setup_code = _read_text(folder / _SETUP_PY, label=folder.name)
|
|
426
|
+
ready_code = _read_text(folder / _READY_PY, label=folder.name)
|
|
427
|
+
setup = _compile_entry(
|
|
428
|
+
setup_code, label=f"{folder.name}/{_SETUP_PY}", entry="setup"
|
|
429
|
+
)
|
|
430
|
+
ready = _compile_entry(
|
|
431
|
+
ready_code, label=f"{folder.name}/{_READY_PY}", entry="ready"
|
|
432
|
+
)
|
|
433
|
+
|
|
434
|
+
sub_goals: list[_CompiledSubGoal] = []
|
|
435
|
+
for name in sub_goal_names:
|
|
436
|
+
check_path = folder / _CHECKS_DIRNAME / f"{name}.py"
|
|
437
|
+
if check_path.is_file():
|
|
438
|
+
check_code = _read_text(check_path, label=folder.name)
|
|
439
|
+
check = _compile_entry(
|
|
440
|
+
check_code,
|
|
441
|
+
label=f"{folder.name}/{_CHECKS_DIRNAME}/{name}.py",
|
|
442
|
+
entry="check",
|
|
443
|
+
allow_empty=False, # R1-2: an existing-but-empty check file is invalid, never a
|
|
444
|
+
# vacuous pass -- absence of the file is what means "judged".
|
|
445
|
+
)
|
|
446
|
+
judged = ""
|
|
447
|
+
elif name in (settled_in_code or set()):
|
|
448
|
+
# The catalogue settles this one in code and the file is not here, so the check was
|
|
449
|
+
# written and never materialised. Reporting it as judged is the expensive wrong answer:
|
|
450
|
+
# the scenario runs, the checkpoint reads as assessed, and nothing measured it.
|
|
451
|
+
raise ScenarioDocumentInvalid(
|
|
452
|
+
f"{folder.name}: sub-goal {name!r} is settled in code by the catalogue but "
|
|
453
|
+
f"{_CHECKS_DIRNAME}/{name}.py is missing, so nothing would measure it"
|
|
454
|
+
)
|
|
455
|
+
else:
|
|
456
|
+
# No `checks/<name>.py` -- per `write_folder`'s own `deterministic()` filter, this
|
|
457
|
+
# name is a JUDGED sub-goal.
|
|
458
|
+
judged = _JUDGED_MARKER
|
|
459
|
+
check = _judged_placeholder_check
|
|
460
|
+
sub_goals.append(_CompiledSubGoal(name=name, judged=judged, check=check))
|
|
461
|
+
|
|
462
|
+
return _CompiledScenario(
|
|
463
|
+
scenario_key=scenario_key,
|
|
464
|
+
scenario_id=scenario_id,
|
|
465
|
+
sub_goals=tuple(sub_goals),
|
|
466
|
+
requires_tool_evidence=requires_tool_evidence,
|
|
467
|
+
setup=setup,
|
|
468
|
+
ready=ready,
|
|
469
|
+
presented=_presented(body, scenario_key=scenario_key),
|
|
470
|
+
)
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _with_claims(
|
|
474
|
+
scenario: _CompiledScenario, claims: dict[str, dict[str, str]]
|
|
475
|
+
) -> _CompiledScenario:
|
|
476
|
+
"""Restore each sub-goal's real claim from the catalogue.
|
|
477
|
+
|
|
478
|
+
`_load_one` can only tell that a sub-goal is judged, never what it was meant to decide:
|
|
479
|
+
`folder.py` writes no file for one. Without this the platform judge gets a name and a
|
|
480
|
+
placeholder, which is not something a verdict can be reached from.
|
|
481
|
+
|
|
482
|
+
`what` is restored for CODED sub-goals too, not only judged ones. A check says nothing when it
|
|
483
|
+
holds, so `what` is the only thing a reader has to tell a real pass from one nobody wrote a
|
|
484
|
+
check for; withholding it left every passing sub-goal explaining itself as "the check found
|
|
485
|
+
nothing wrong". `judged` stays restricted to judged sub-goals, since a coded one has no claim
|
|
486
|
+
for a model to decide.
|
|
487
|
+
"""
|
|
488
|
+
if not claims:
|
|
489
|
+
return scenario
|
|
490
|
+
restored = tuple(
|
|
491
|
+
replace(
|
|
492
|
+
goal,
|
|
493
|
+
judged=(claims[goal.name].get("judged") or goal.judged)
|
|
494
|
+
if goal.judged
|
|
495
|
+
else goal.judged,
|
|
496
|
+
what=claims[goal.name].get("what", "") or goal.what,
|
|
497
|
+
)
|
|
498
|
+
if goal.name in claims
|
|
499
|
+
else goal
|
|
500
|
+
for goal in scenario.sub_goals
|
|
501
|
+
)
|
|
502
|
+
return replace(scenario, sub_goals=restored)
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def load_scenarios(bundle_dir: Path) -> list[_CompiledScenario]:
|
|
506
|
+
"""Every scenario document under `<bundle_dir>/scenarios/`, compiled and wrapped, in the same
|
|
507
|
+
sorted-by-folder-name order `folder.py`'s `read_all` uses. Raises `ScenarioDocumentInvalid` on
|
|
508
|
+
the FIRST unreadable or malformed folder -- unlike `read_all`, which skips one and continues;
|
|
509
|
+
a hosted job has nobody watching a suite by hand to notice a scenario silently missing from the
|
|
510
|
+
count, so a folder this reader cannot use fails the whole job instead of shrinking it quietly.
|
|
511
|
+
"""
|
|
512
|
+
root = bundle_dir / SCENARIOS_DIRNAME
|
|
513
|
+
if not root.is_dir():
|
|
514
|
+
raise ScenarioDocumentInvalid(f"{root} is not a directory")
|
|
515
|
+
try:
|
|
516
|
+
entries = sorted(root.iterdir())
|
|
517
|
+
except OSError as exc:
|
|
518
|
+
# An unreadable `scenarios/` directory is the same typed failure as any other malformed
|
|
519
|
+
# document (R1-1) -- this is inside `run_job`'s `try`/`except ScenarioDocumentInvalid`
|
|
520
|
+
# (unlike `bundle_has_scenarios`'s own guard above), so raising here is the safe direction.
|
|
521
|
+
raise ScenarioDocumentInvalid(
|
|
522
|
+
f"{root}: cannot list scenario folders: {exc}"
|
|
523
|
+
) from exc
|
|
524
|
+
settled_in_code = _deterministic_names(bundle_dir)
|
|
525
|
+
declared_tools = _declared_tool_names(bundle_dir)
|
|
526
|
+
claims = _load_catalogue_claims(bundle_dir)
|
|
527
|
+
scenarios: list[_CompiledScenario] = []
|
|
528
|
+
for folder in entries:
|
|
529
|
+
if not folder.is_dir():
|
|
530
|
+
continue
|
|
531
|
+
scenarios.append(
|
|
532
|
+
_with_claims(
|
|
533
|
+
_load_one(
|
|
534
|
+
folder,
|
|
535
|
+
settled_in_code=settled_in_code,
|
|
536
|
+
declared_tools=declared_tools,
|
|
537
|
+
),
|
|
538
|
+
claims,
|
|
539
|
+
)
|
|
540
|
+
)
|
|
541
|
+
if not scenarios:
|
|
542
|
+
raise ScenarioDocumentInvalid(f"{root} contains no scenario folders")
|
|
543
|
+
return scenarios
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
class BundleScenarioSource:
|
|
547
|
+
"""The real `ScenarioSource`: reads and compiles the bundle's own scenario documents.
|
|
548
|
+
`hosted_entrypoint.run_job` wires this in only when the injected source is still the default
|
|
549
|
+
`NotWiredScenarioSource` AND the bundle actually carries a `scenarios/` directory (the LAYOUT
|
|
550
|
+
DECISION's presence test) -- an injected `ScenarioSource` (every test, every future caller)
|
|
551
|
+
always wins over this one.
|
|
552
|
+
"""
|
|
553
|
+
|
|
554
|
+
async def build(
|
|
555
|
+
self,
|
|
556
|
+
job: Any,
|
|
557
|
+
bundle: Any,
|
|
558
|
+
scenarios_client: "ScenariosClient",
|
|
559
|
+
*,
|
|
560
|
+
pool: Any,
|
|
561
|
+
world_factory: Any,
|
|
562
|
+
bundle_dir: Path,
|
|
563
|
+
) -> Sequence[_CompiledScenario]:
|
|
564
|
+
del bundle, pool, world_factory
|
|
565
|
+
# `Path.read_text`/`iterdir`/`compile` are all blocking filesystem+CPU work -- run off the
|
|
566
|
+
# event loop the same way `hosted_entrypoint.py` already does for `bundle_source.load` and
|
|
567
|
+
# `preflight_bundle`, rather than stalling every other in-flight scenario behind it.
|
|
568
|
+
try:
|
|
569
|
+
scenarios = await asyncio.wait_for(
|
|
570
|
+
asyncio.to_thread(load_scenarios, bundle_dir),
|
|
571
|
+
timeout=_LOAD_TIMEOUT_SECONDS,
|
|
572
|
+
)
|
|
573
|
+
except asyncio.TimeoutError as exc:
|
|
574
|
+
# R1-5: the underlying thread cannot actually be canceled/killed -- it is left running
|
|
575
|
+
# in the background. Converting the hang into a typed terminal here is still strictly
|
|
576
|
+
# better than the pre-fix behavior (no terminal event, ever): the job gets an honest,
|
|
577
|
+
# bounded FAILED verdict instead of hanging until the platform's own wall clock gives up.
|
|
578
|
+
raise ScenarioDocumentInvalid(
|
|
579
|
+
f"{bundle_dir / SCENARIOS_DIRNAME}: loading scenario documents exceeded "
|
|
580
|
+
f"{_LOAD_TIMEOUT_SECONDS:.0f}s"
|
|
581
|
+
) from exc
|
|
582
|
+
# An empty `scenario_key` is carried VERBATIM off the document by design (module docstring
|
|
583
|
+
# -- never synthesized here), but it is also the one shape `hosted_entrypoint.py`'s own
|
|
584
|
+
# downstream `_validate_scenarios` would reject as a local, deterministic ENVIRONMENT-domain
|
|
585
|
+
# content defect -- checked HERE, before `register_with_platform` ever reaches the network,
|
|
586
|
+
# so that cheaper, existing local classification wins over a round trip that would only
|
|
587
|
+
# rediscover the same defect as a `platform_sync` failure instead (Azain's serializer
|
|
588
|
+
# rejects a blank `scenario_key` with its own 400 -- `scenario_key` is a plain
|
|
589
|
+
# non-`allow_blank` `CharField`, hosted_harness.py:169). `ScenarioDocumentInvalid` reuses
|
|
590
|
+
# `run_job`'s EXISTING `except ScenarioDocumentInvalid` clause (domain=environment) --
|
|
591
|
+
# nothing new to catch there.
|
|
592
|
+
if any(not scenario.scenario_key for scenario in scenarios):
|
|
593
|
+
raise ScenarioDocumentInvalid(
|
|
594
|
+
f"{bundle_dir / SCENARIOS_DIRNAME}: a scenario document has no non-empty "
|
|
595
|
+
"scenario_key"
|
|
596
|
+
)
|
|
597
|
+
# p13: pre-allocation, after load and before the scheduler ever sees a scenario (spine
|
|
598
|
+
# step 3.5) -- `register_with_platform` raises `ScenarioPreallocationError`/
|
|
599
|
+
# `ob.HostedFencedError`/`ob.HostedChannelFailedError`/`ob.HostedAttemptSupersededError` on
|
|
600
|
+
# any failure, all of which `hosted_entrypoint.run_job`'s existing call site around
|
|
601
|
+
# `scenario_source.build()` already maps to the typed `validating_scenarios`/`platform_sync`
|
|
602
|
+
# terminal (or the fenced exit) -- nothing new to catch here.
|
|
603
|
+
# Pre-allocate the WHOLE suite, then run a sample of it.
|
|
604
|
+
#
|
|
605
|
+
# The sample used to be taken first, so the platform was handed five personas for a suite of
|
|
606
|
+
# thirty and refused the job: "expected exactly 30 personas, got 5". Pre-allocation is sealed
|
|
607
|
+
# against the full set by design (`_begin_payload` sends every key and the platform 409s on a
|
|
608
|
+
# subset), so the suite is what gets registered and the sample is only what gets called. The
|
|
609
|
+
# rows that are not called stay unstarted, which is a truthful state rather than a broken job.
|
|
610
|
+
chosen_evals, agent_prompt, modality = _chosen_evals_and_prompt(bundle_dir)
|
|
611
|
+
run_name = _derive_run_name(job, bundle_dir)
|
|
612
|
+
agent_name = _derive_agent_name(job, bundle_dir)
|
|
613
|
+
registered = await register_with_platform(
|
|
614
|
+
scenarios_client,
|
|
615
|
+
scenarios,
|
|
616
|
+
run_name=run_name,
|
|
617
|
+
agent_name=agent_name,
|
|
618
|
+
chosen_evals=chosen_evals,
|
|
619
|
+
agent_prompt=agent_prompt,
|
|
620
|
+
modality=modality,
|
|
621
|
+
)
|
|
622
|
+
return sampled_for_calling(registered)
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _chosen_evals_and_prompt(bundle_dir: Path) -> tuple[list[str], str, str]:
|
|
626
|
+
"""The contract's chosen evals, agent prompt and modality; read as plain JSON so it cannot fail a run."""
|
|
627
|
+
try:
|
|
628
|
+
body = json.loads((bundle_dir / "contract.json").read_text(encoding="utf-8"))
|
|
629
|
+
except Exception: # noqa: BLE001 - a run never fails over what it tells the platform about itself
|
|
630
|
+
return [], "", ""
|
|
631
|
+
if not isinstance(body, dict):
|
|
632
|
+
return [], "", ""
|
|
633
|
+
chosen = body.get("chosen_evals")
|
|
634
|
+
names = (
|
|
635
|
+
[str(one).strip() for one in chosen if str(one).strip()]
|
|
636
|
+
if isinstance(chosen, list)
|
|
637
|
+
else []
|
|
638
|
+
)
|
|
639
|
+
# Provisioning defaults to text, which would bind a voice run's evals to the transcript.
|
|
640
|
+
modality = str(body.get("modality") or "").strip().lower()
|
|
641
|
+
return (
|
|
642
|
+
names,
|
|
643
|
+
str(body.get("system_prompt_excerpt") or "").strip(),
|
|
644
|
+
modality if modality in ("voice", "text") else "",
|
|
645
|
+
)
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _preallocation_error(code: str, message: str) -> Exception:
|
|
649
|
+
"""Builds a `hosted_entrypoint.ScenarioPreallocationError` for a guard failure below --
|
|
650
|
+
imported lazily (not at module level) because `hosted_entrypoint.py` imports THIS module at
|
|
651
|
+
its own top level (`BundleScenarioSource`/`ScenarioDocumentInvalid`/`bundle_has_scenarios`), so
|
|
652
|
+
a top-level import back would be a circular import. Reusing that exact exception class (rather
|
|
653
|
+
than inventing a new one) is what lets these guard failures land on `run_job`'s ALREADY-WIRED
|
|
654
|
+
`except (ScenarioSourceNotWired, ScenarioPreallocationError)` clause with no changes there.
|
|
655
|
+
"""
|
|
656
|
+
from .hosted_entrypoint import ScenarioPreallocationError
|
|
657
|
+
|
|
658
|
+
return ScenarioPreallocationError(
|
|
659
|
+
ob.ChannelError(
|
|
660
|
+
ob.ChannelOutcome.PERMANENT_ITEM, FailureDomain.PLATFORM_SYNC, code, message
|
|
661
|
+
)
|
|
662
|
+
)
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
def _read_bundle_contract(bundle_dir: Path) -> dict[str, Any]:
|
|
666
|
+
"""Best-effort read of the authored contract from the bundle directory."""
|
|
667
|
+
path = bundle_dir / "contract.json"
|
|
668
|
+
if not path.is_file():
|
|
669
|
+
return {}
|
|
670
|
+
try:
|
|
671
|
+
body = json.loads(path.read_text(encoding="utf-8"))
|
|
672
|
+
return body if isinstance(body, dict) else {}
|
|
673
|
+
except (OSError, ValueError):
|
|
674
|
+
return {}
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
def _derive_run_name(job: Any, bundle_dir: Path) -> str:
|
|
678
|
+
"""Human-readable simulation run name from the contract or job source.
|
|
679
|
+
|
|
680
|
+
Prefer the authored contract's ``agent`` field (e.g.
|
|
681
|
+
``uber_voice_agent`` -> ``Uber Voice Agent``); fall back to the last
|
|
682
|
+
path segment of ``source.repository`` (e.g.
|
|
683
|
+
``future-agi/ride-voice-agent`` -> ``ride-voice-agent``).
|
|
684
|
+
"""
|
|
685
|
+
contract = _read_bundle_contract(bundle_dir)
|
|
686
|
+
agent = str(contract.get("agent") or "").strip()
|
|
687
|
+
if agent:
|
|
688
|
+
return agent.replace("_", " ").replace("-", " ").title()[:200]
|
|
689
|
+
|
|
690
|
+
repo = getattr(getattr(job, "source", None), "repository", None) or ""
|
|
691
|
+
if "/" in repo:
|
|
692
|
+
return repo.rsplit("/", 1)[-1][:200]
|
|
693
|
+
if repo:
|
|
694
|
+
return repo[:200]
|
|
695
|
+
return "simulation"
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
def _derive_agent_name(job: Any, bundle_dir: Path) -> str:
|
|
699
|
+
"""Agent name for the provision payload, from contract or source repo."""
|
|
700
|
+
contract = _read_bundle_contract(bundle_dir)
|
|
701
|
+
agent = str(contract.get("agent") or "").strip()
|
|
702
|
+
if agent:
|
|
703
|
+
return agent.replace("_", " ").replace("-", " ").title()[:200]
|
|
704
|
+
|
|
705
|
+
repo = getattr(getattr(job, "source", None), "repository", None) or ""
|
|
706
|
+
if "/" in repo:
|
|
707
|
+
return repo.rsplit("/", 1)[-1][:200]
|
|
708
|
+
if repo:
|
|
709
|
+
return repo[:200]
|
|
710
|
+
return "alk-agent"
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
def _provision_payload(
|
|
714
|
+
run_name: str,
|
|
715
|
+
scenarios: Sequence[_CompiledScenario],
|
|
716
|
+
chosen_evals: Sequence[str] = (),
|
|
717
|
+
agent_prompt: str = "",
|
|
718
|
+
modality: str = "",
|
|
719
|
+
*,
|
|
720
|
+
agent_name: str = "",
|
|
721
|
+
) -> dict[str, Any]:
|
|
722
|
+
"""`HarnessScenarioProvisionSerializer`/`HarnessProvisionPersonaSerializer`
|
|
723
|
+
(futureagi/simulate/serializers/hosted_harness.py): `operation`, `name` and `personas` are
|
|
724
|
+
required; each persona's `name`, `role`, `situation`, `outcome` and `persona` are optional and
|
|
725
|
+
are now supplied from the document (`_presented`). They were omitted while this module read
|
|
726
|
+
`scenario.json` for scheduler-facing fields only, and the cost was a platform that could show a
|
|
727
|
+
call but not who was on it or what they came for.
|
|
728
|
+
"""
|
|
729
|
+
payload: dict[str, Any] = {
|
|
730
|
+
"operation": "provision",
|
|
731
|
+
"name": run_name,
|
|
732
|
+
# Everything the serializer accepts, where the document had it: name, role, situation,
|
|
733
|
+
# outcome and the persona itself. `scenario_key` is set last because it is the one field the
|
|
734
|
+
# platform matches on and it must be this scenario's, whatever the persona record says.
|
|
735
|
+
"personas": [
|
|
736
|
+
{**scenario.presented, "scenario_key": scenario.scenario_key}
|
|
737
|
+
for scenario in scenarios
|
|
738
|
+
],
|
|
739
|
+
}
|
|
740
|
+
if agent_name:
|
|
741
|
+
payload["agent_name"] = agent_name
|
|
742
|
+
# Each omitted rather than sent empty, so an older platform is unaffected. Modality matters
|
|
743
|
+
# because provisioning defaults to text, which binds every eval to the transcript.
|
|
744
|
+
if chosen_evals:
|
|
745
|
+
payload["chosen_evals"] = list(chosen_evals)
|
|
746
|
+
if agent_prompt:
|
|
747
|
+
payload["agent_prompt"] = agent_prompt
|
|
748
|
+
if modality:
|
|
749
|
+
payload["modality"] = modality
|
|
750
|
+
return payload
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
def _begin_payload(
|
|
754
|
+
run_test_id: str, scenarios: Sequence[_CompiledScenario]
|
|
755
|
+
) -> dict[str, Any]:
|
|
756
|
+
"""`HarnessScenarioBeginSerializer` (futureagi/simulate/serializers/hosted_harness.py:193-198):
|
|
757
|
+
`scenario_keys` is `allow_empty=False` and REQUIRED, and `begin_scenarios`
|
|
758
|
+
(services/hosted_harness.py:323-329) 409s (`scenario_key_mismatch`) on anything but an EXACT
|
|
759
|
+
match against the full sealed set -- there is no "subset to run" semantics on the real
|
|
760
|
+
platform (that contract text describes an optional partial-subset `scenario_ids`; the
|
|
761
|
+
live route does not implement that -- CONTRACT NOTES). The full set is sent every time.
|
|
762
|
+
"""
|
|
763
|
+
return {
|
|
764
|
+
"operation": "begin",
|
|
765
|
+
"run_test_id": run_test_id,
|
|
766
|
+
"scenario_keys": [scenario.scenario_key for scenario in scenarios],
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
|
|
770
|
+
def _scenario_ids_by_key(
|
|
771
|
+
submitted: Sequence[_CompiledScenario], raw_scenarios: Any
|
|
772
|
+
) -> dict[str, str]:
|
|
773
|
+
"""Matches the platform's KEYED provision response
|
|
774
|
+
(`{"scenarios": [{"scenario_key", "scenario_id"}, ...]}`,
|
|
775
|
+
futureagi/simulate/serializers/hosted_harness.py:251-260 +
|
|
776
|
+
services/hosted_harness.py:487-501's `_provision_response`) back onto `submitted` BY
|
|
777
|
+
`scenario_key` -- a dict lookup, never a positional zip. A positional zip (matching that contract's
|
|
778
|
+
documented `scenario_ids` array shape, not what the platform actually returns) would silently
|
|
779
|
+
mismatch scenario_id -> scenario the instant the response order differs from `submitted`'s
|
|
780
|
+
order, which nothing on the wire guarantees. Every check below raises rather than returning a
|
|
781
|
+
partial mapping -- "never partial assignment" per the brief: the caller only gets a mapping
|
|
782
|
+
once it is proven complete (every submitted key present, exactly once) and exact (no
|
|
783
|
+
unrecognized key).
|
|
784
|
+
"""
|
|
785
|
+
if not isinstance(raw_scenarios, list):
|
|
786
|
+
raise _preallocation_error(
|
|
787
|
+
"scenarios_provision_response_invalid", "response 'scenarios' is not a list"
|
|
788
|
+
)
|
|
789
|
+
by_key: dict[str, str] = {}
|
|
790
|
+
for entry in raw_scenarios:
|
|
791
|
+
if not isinstance(entry, dict):
|
|
792
|
+
raise _preallocation_error(
|
|
793
|
+
"scenarios_provision_response_invalid",
|
|
794
|
+
"a 'scenarios' entry is not an object",
|
|
795
|
+
)
|
|
796
|
+
key = entry.get("scenario_key")
|
|
797
|
+
scenario_id = entry.get("scenario_id")
|
|
798
|
+
if not isinstance(key, str) or not key:
|
|
799
|
+
raise _preallocation_error(
|
|
800
|
+
"scenarios_provision_response_invalid",
|
|
801
|
+
"a 'scenarios' entry has no non-empty scenario_key",
|
|
802
|
+
)
|
|
803
|
+
if key in by_key:
|
|
804
|
+
raise _preallocation_error(
|
|
805
|
+
"scenario_registration_duplicate_key",
|
|
806
|
+
f"scenario_key {key!r} appears more than once in the provision response",
|
|
807
|
+
)
|
|
808
|
+
if not isinstance(scenario_id, str) or not scenario_id:
|
|
809
|
+
raise _preallocation_error(
|
|
810
|
+
"scenarios_provision_response_invalid",
|
|
811
|
+
f"scenario_key {key!r} has no non-empty scenario_id",
|
|
812
|
+
)
|
|
813
|
+
by_key[key] = scenario_id
|
|
814
|
+
|
|
815
|
+
submitted_keys = [scenario.scenario_key for scenario in submitted]
|
|
816
|
+
unknown = sorted(set(by_key) - set(submitted_keys))
|
|
817
|
+
if unknown:
|
|
818
|
+
raise _preallocation_error(
|
|
819
|
+
"scenario_registration_unknown_key",
|
|
820
|
+
f"provision response named scenario_key(s) never submitted: {unknown}",
|
|
821
|
+
)
|
|
822
|
+
missing = sorted(set(submitted_keys) - set(by_key))
|
|
823
|
+
if missing:
|
|
824
|
+
raise _preallocation_error(
|
|
825
|
+
"scenario_registration_missing",
|
|
826
|
+
f"provision response is missing scenario_key(s): {missing}",
|
|
827
|
+
)
|
|
828
|
+
return by_key
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
async def register_with_platform(
|
|
832
|
+
scenarios_client: "ScenariosClient",
|
|
833
|
+
scenarios: Sequence[_CompiledScenario],
|
|
834
|
+
*,
|
|
835
|
+
run_name: str,
|
|
836
|
+
agent_name: str = "",
|
|
837
|
+
chosen_evals: Sequence[str] = (),
|
|
838
|
+
agent_prompt: str = "",
|
|
839
|
+
modality: str = "",
|
|
840
|
+
) -> Sequence[_CompiledScenario]:
|
|
841
|
+
"""The scenario pre-allocation SEAM, now wired against the platform's real route (a single
|
|
842
|
+
`POST .../scenarios/`, discriminated by a body-level `operation` field -- see
|
|
843
|
+
`ScenariosClient`'s own docstring for the file:line evidence). `.provision()`/`.begin()` are
|
|
844
|
+
blocking network calls (same `ScenariosClient` the rest of `hosted_entrypoint.py` already
|
|
845
|
+
drives off the event loop via `asyncio.to_thread` -- matched here rather than diverging).
|
|
846
|
+
|
|
847
|
+
Sequence: provision (get platform-assigned ids, keyed by `scenario_key`) -> match ids back
|
|
848
|
+
onto `scenarios` with hard guards (`_scenario_ids_by_key`, raises before ANY assignment on any
|
|
849
|
+
mismatch) -> begin (seals execution against the FULL scenario_keys set; a begin failure means
|
|
850
|
+
NO scenario in this batch is returned with an id -- the whole call raises, same as a provision
|
|
851
|
+
failure) -> only then build and return the new scenario list with `scenario_id` filled in.
|
|
852
|
+
"""
|
|
853
|
+
provision_result = await asyncio.to_thread(
|
|
854
|
+
scenarios_client.provision,
|
|
855
|
+
_provision_payload(
|
|
856
|
+
run_name,
|
|
857
|
+
scenarios,
|
|
858
|
+
chosen_evals,
|
|
859
|
+
agent_prompt,
|
|
860
|
+
modality,
|
|
861
|
+
agent_name=agent_name,
|
|
862
|
+
),
|
|
863
|
+
)
|
|
864
|
+
run_test_id = provision_result.get("run_test_id")
|
|
865
|
+
if not isinstance(run_test_id, str) or not run_test_id:
|
|
866
|
+
raise _preallocation_error(
|
|
867
|
+
"scenarios_provision_response_invalid",
|
|
868
|
+
"provision response has no run_test_id",
|
|
869
|
+
)
|
|
870
|
+
id_by_key = _scenario_ids_by_key(scenarios, provision_result.get("scenarios"))
|
|
871
|
+
|
|
872
|
+
await asyncio.to_thread(
|
|
873
|
+
scenarios_client.begin, _begin_payload(run_test_id, scenarios)
|
|
874
|
+
)
|
|
875
|
+
|
|
876
|
+
return tuple(
|
|
877
|
+
replace(scenario, scenario_id=id_by_key[scenario.scenario_key])
|
|
878
|
+
for scenario in scenarios
|
|
879
|
+
)
|