agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/harness/cli.py
ADDED
|
@@ -0,0 +1,1354 @@
|
|
|
1
|
+
"""Run a stage from a terminal.
|
|
2
|
+
|
|
3
|
+
This is one renderer over the stage loop, not the product. It prints events as lines and reads
|
|
4
|
+
follow-ups from stdin; a browser front end subscribes to the same events and draws them as a
|
|
5
|
+
transcript beside the artifact. Keeping the terminal a renderer rather than the interface is what
|
|
6
|
+
makes the second one cheap.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import asyncio
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
import time
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from . import spend
|
|
21
|
+
from . import observability
|
|
22
|
+
from .build import open_stage as build_stage
|
|
23
|
+
from .build import opening as build_opening
|
|
24
|
+
from .build import require_buildable
|
|
25
|
+
from .chat import open_conversation
|
|
26
|
+
from .config import (
|
|
27
|
+
artifact_dir,
|
|
28
|
+
chosen_model,
|
|
29
|
+
credentials_hint,
|
|
30
|
+
permission_gate,
|
|
31
|
+
)
|
|
32
|
+
from .run.targets import supported as target_kinds
|
|
33
|
+
from .scenarios import load as load_written
|
|
34
|
+
from .scenarios import open_stage as scenario_stage
|
|
35
|
+
from .scenarios import opening as scenario_opening
|
|
36
|
+
from .session import TEXT, Event
|
|
37
|
+
from .sessions import Session, new_id, save as save_session
|
|
38
|
+
from .sources import resolve, supported
|
|
39
|
+
from .understand import load, open_stage, opening
|
|
40
|
+
from .world.snapshot import saved as world_saved
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _source_root(destination: Path, explicit: str = "") -> str:
|
|
44
|
+
"""Recover the source path for commands resumed from a session folder."""
|
|
45
|
+
if explicit.strip():
|
|
46
|
+
return str(Path(explicit).expanduser().resolve())
|
|
47
|
+
metadata = destination / "session.json"
|
|
48
|
+
if metadata.exists():
|
|
49
|
+
try:
|
|
50
|
+
import json
|
|
51
|
+
|
|
52
|
+
return str(
|
|
53
|
+
json.loads(metadata.read_text(encoding="utf-8")).get("source") or ""
|
|
54
|
+
)
|
|
55
|
+
except (OSError, ValueError):
|
|
56
|
+
pass
|
|
57
|
+
return ""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _render(event: Event) -> None:
|
|
61
|
+
line = event.line()
|
|
62
|
+
if event.kind == TEXT:
|
|
63
|
+
print(line, end="", flush=True)
|
|
64
|
+
else:
|
|
65
|
+
print(f"\n{line}", flush=True)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
async def _prompt(question: str) -> str:
|
|
69
|
+
return (await asyncio.to_thread(input, question)).strip()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
async def _ask_operator(_tool_name: str, payload: dict[str, Any], _context: Any) -> Any:
|
|
73
|
+
"""Render the model's clarifying questions and return the operator's answers."""
|
|
74
|
+
from claude_agent_sdk.types import PermissionResultAllow
|
|
75
|
+
|
|
76
|
+
answers: dict[str, Any] = {}
|
|
77
|
+
for question in payload.get("questions", []):
|
|
78
|
+
print(f"\n\n {question.get('header', '?')}: {question.get('question', '')}")
|
|
79
|
+
options = question.get("options", []) or []
|
|
80
|
+
for index, option in enumerate(options, start=1):
|
|
81
|
+
print(
|
|
82
|
+
f" {index}. {option.get('label')} - {option.get('description', '')}"
|
|
83
|
+
)
|
|
84
|
+
raw = await _prompt(" > ")
|
|
85
|
+
chosen = raw
|
|
86
|
+
if raw.isdigit() and 1 <= int(raw) <= len(options):
|
|
87
|
+
chosen = options[int(raw) - 1].get("label", raw)
|
|
88
|
+
answers[question.get("question", "")] = chosen
|
|
89
|
+
print()
|
|
90
|
+
return PermissionResultAllow(
|
|
91
|
+
updated_input={"questions": payload.get("questions", []), "answers": answers}
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _guidance(args: argparse.Namespace) -> str:
|
|
96
|
+
instructions = [
|
|
97
|
+
str(item).strip()
|
|
98
|
+
for item in (getattr(args, "guidance", None) or [])
|
|
99
|
+
if str(item).strip()
|
|
100
|
+
]
|
|
101
|
+
if not instructions:
|
|
102
|
+
return ""
|
|
103
|
+
return (
|
|
104
|
+
"\n\n## User adjustments (required completion criteria)\n\n"
|
|
105
|
+
"The user supplied the requirements below during this run. The final saved "
|
|
106
|
+
"artifact MUST directly represent every bullet; acknowledging a bullet or merely "
|
|
107
|
+
"changing the requested count is not enough. For scenario-stage adjustments, at "
|
|
108
|
+
"least one saved scenario must clearly test each requested behavior. Preserve all "
|
|
109
|
+
"unaffected validated work. Do not call save_scenarios until these requirements are "
|
|
110
|
+
"visible in the saved suite:\n- " + "\n- ".join(instructions)
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _scenario_adjustment_requirement(adjustment_id: str, instruction: str) -> str:
|
|
115
|
+
"""Make a live scenario correction mechanically auditable after generation."""
|
|
116
|
+
return (
|
|
117
|
+
f"Adjustment {adjustment_id}: {instruction}\n"
|
|
118
|
+
"At least one scenario that directly tests this requirement MUST include "
|
|
119
|
+
f'fixture.adjustment_ids containing "{adjustment_id}". This marker is required '
|
|
120
|
+
"evidence that the saved suite reflects the instruction."
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _missing_scenario_adjustments(
|
|
125
|
+
written: list[Any], adjustment_ids: list[str]
|
|
126
|
+
) -> list[str]:
|
|
127
|
+
covered: set[str] = set()
|
|
128
|
+
for scenario in written:
|
|
129
|
+
fixture = getattr(scenario, "fixture", None)
|
|
130
|
+
if not isinstance(fixture, dict):
|
|
131
|
+
continue
|
|
132
|
+
markers = fixture.get("adjustment_ids") or []
|
|
133
|
+
if isinstance(markers, str):
|
|
134
|
+
markers = [markers]
|
|
135
|
+
if isinstance(markers, list):
|
|
136
|
+
covered.update(str(marker) for marker in markers)
|
|
137
|
+
return [
|
|
138
|
+
adjustment_id
|
|
139
|
+
for adjustment_id in adjustment_ids
|
|
140
|
+
if adjustment_id not in covered
|
|
141
|
+
]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
async def _understand(args: argparse.Namespace) -> int:
|
|
145
|
+
if args.kind == "provider":
|
|
146
|
+
source = resolve(
|
|
147
|
+
args.kind,
|
|
148
|
+
name=args.name,
|
|
149
|
+
profile=getattr(args, "provider_profile", None) or {},
|
|
150
|
+
scratch=args.path,
|
|
151
|
+
)
|
|
152
|
+
else:
|
|
153
|
+
source = resolve(args.kind, name=args.name, root=args.path)
|
|
154
|
+
job = getattr(args, "job", None)
|
|
155
|
+
offered = (
|
|
156
|
+
(getattr(job, "metadata", None) or {}).get("available_evals") if job else None
|
|
157
|
+
)
|
|
158
|
+
stage, destination = open_stage(
|
|
159
|
+
source,
|
|
160
|
+
out=Path(args.out) if args.out else None,
|
|
161
|
+
# Unattended, there is nobody to answer, so the model records what it could not
|
|
162
|
+
# resolve in open_questions rather than blocking on a prompt nobody will see.
|
|
163
|
+
ask=permission_gate(_ask_operator) if args.interactive else None,
|
|
164
|
+
available_evals=offered if isinstance(offered, list) else None,
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
print(f"agent: {source.name} ({source.kind})")
|
|
168
|
+
print(f"model: {chosen_model()}")
|
|
169
|
+
print(f"out: {destination}\n")
|
|
170
|
+
|
|
171
|
+
await _converse(
|
|
172
|
+
stage,
|
|
173
|
+
opening(source) + _guidance(args),
|
|
174
|
+
interactive=args.interactive,
|
|
175
|
+
until=lambda: load(destination) is not None,
|
|
176
|
+
nudge=(
|
|
177
|
+
"Nothing was saved: you finished without calling submit_contract. Call it now "
|
|
178
|
+
"with the contract you worked out."
|
|
179
|
+
),
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
contract = load(destination)
|
|
183
|
+
if contract is None:
|
|
184
|
+
print("\nNo contract was submitted.", file=sys.stderr)
|
|
185
|
+
return 1
|
|
186
|
+
print(
|
|
187
|
+
f"\ncontract: {len(contract.tools)} tools, "
|
|
188
|
+
f"{len(contract.hard_constraints)} rules, "
|
|
189
|
+
f"{len(contract.real_use_cases)} use cases, "
|
|
190
|
+
f"{len(contract.open_questions)} open questions"
|
|
191
|
+
)
|
|
192
|
+
print(f"spent: ${stage.spent_usd:.4f}")
|
|
193
|
+
return 0
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
async def _converse(
|
|
197
|
+
stage,
|
|
198
|
+
opening_message: str,
|
|
199
|
+
*,
|
|
200
|
+
interactive: bool,
|
|
201
|
+
until=None,
|
|
202
|
+
nudge: str = "",
|
|
203
|
+
) -> None:
|
|
204
|
+
"""Say the opening, then keep the stage open for corrections.
|
|
205
|
+
|
|
206
|
+
The same shape for every stage. A world is usually right on the second look, and the point
|
|
207
|
+
of holding the session open is that correcting it is the next thing said rather than a
|
|
208
|
+
rebuild from nothing.
|
|
209
|
+
|
|
210
|
+
``until``/``nudge`` guard the unattended case. The commonest way an unattended stage fails
|
|
211
|
+
is finishing all the work and never calling the tool that saves it — the whole contract
|
|
212
|
+
written out as prose, submitted to nobody. One mechanical reminder costs a turn; rerunning
|
|
213
|
+
the stage costs everything it just did.
|
|
214
|
+
"""
|
|
215
|
+
async with stage:
|
|
216
|
+
await stage.say(opening_message, on_event=_render)
|
|
217
|
+
if not interactive and until is not None and nudge and not until():
|
|
218
|
+
await stage.say(nudge, on_event=_render)
|
|
219
|
+
while interactive:
|
|
220
|
+
try:
|
|
221
|
+
said = await _prompt("\nyou ")
|
|
222
|
+
except (EOFError, KeyboardInterrupt):
|
|
223
|
+
break
|
|
224
|
+
if not said or said in {"q", "quit", "exit"}:
|
|
225
|
+
break
|
|
226
|
+
await stage.say(said, on_event=_render)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
async def _build(args: argparse.Namespace) -> int:
|
|
230
|
+
destination = Path(args.out) if args.out else artifact_dir(args.name)
|
|
231
|
+
contract = load(destination)
|
|
232
|
+
if contract is None:
|
|
233
|
+
print(f"No contract at {destination}. Run `understand` first.", file=sys.stderr)
|
|
234
|
+
return 1
|
|
235
|
+
|
|
236
|
+
print(f"agent: {contract.agent} ({len(contract.tools)} tools)")
|
|
237
|
+
print(f"model: {chosen_model()}")
|
|
238
|
+
print(f"out: {destination}\n")
|
|
239
|
+
|
|
240
|
+
source_root = _source_root(destination, args.path or "")
|
|
241
|
+
external_runtime = bool(getattr(args, "external_runtime", False))
|
|
242
|
+
try:
|
|
243
|
+
require_buildable(
|
|
244
|
+
contract,
|
|
245
|
+
source_root,
|
|
246
|
+
external_runtime=external_runtime,
|
|
247
|
+
)
|
|
248
|
+
except RuntimeError as failed:
|
|
249
|
+
print(str(failed), file=sys.stderr)
|
|
250
|
+
return 1
|
|
251
|
+
|
|
252
|
+
from .provision import ProvisionError, provision_if_present
|
|
253
|
+
|
|
254
|
+
environment = None
|
|
255
|
+
if not bool(getattr(args, "skip_source_provision", False)):
|
|
256
|
+
try:
|
|
257
|
+
environment = await asyncio.to_thread(
|
|
258
|
+
provision_if_present, source_root, destination, contract
|
|
259
|
+
)
|
|
260
|
+
except ProvisionError as failed:
|
|
261
|
+
print(f"Cannot create the source environment: {failed}", file=sys.stderr)
|
|
262
|
+
return 1
|
|
263
|
+
if environment is not None:
|
|
264
|
+
print(f"environment: {environment.project} ({', '.join(environment.services)})")
|
|
265
|
+
for name, value in sorted(environment.overrides.items()):
|
|
266
|
+
print(f"override: {name}={value}")
|
|
267
|
+
|
|
268
|
+
stage, _ = build_stage(
|
|
269
|
+
contract,
|
|
270
|
+
out=destination,
|
|
271
|
+
ask=permission_gate(_ask_operator) if args.interactive else None,
|
|
272
|
+
source_root=source_root,
|
|
273
|
+
deferred_runtime=bool(getattr(args, "skip_source_provision", False)),
|
|
274
|
+
external_runtime=external_runtime,
|
|
275
|
+
)
|
|
276
|
+
deferred_runtime = bool(getattr(args, "skip_source_provision", False))
|
|
277
|
+
await _converse(
|
|
278
|
+
stage,
|
|
279
|
+
build_opening(
|
|
280
|
+
contract,
|
|
281
|
+
provisioned=environment is not None,
|
|
282
|
+
deferred_runtime=deferred_runtime,
|
|
283
|
+
external_runtime=external_runtime,
|
|
284
|
+
)
|
|
285
|
+
+ _guidance(args),
|
|
286
|
+
interactive=args.interactive,
|
|
287
|
+
until=lambda: world_saved(destination),
|
|
288
|
+
nudge=(
|
|
289
|
+
"Nothing was saved: you finished without calling save_world. Call check_world, "
|
|
290
|
+
"fix what it names, then save_world."
|
|
291
|
+
),
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
if not world_saved(destination):
|
|
295
|
+
print("\nNo world was saved.", file=sys.stderr)
|
|
296
|
+
return 1
|
|
297
|
+
# Seal the exact environment now that its generated world exists. Local and hosted
|
|
298
|
+
# execution consume this same internal bundle; the source repository is never a special
|
|
299
|
+
# runtime path after this boundary.
|
|
300
|
+
from .bundle import BundleError, export_session_bundle
|
|
301
|
+
|
|
302
|
+
try:
|
|
303
|
+
bundle_path, bundle = await asyncio.to_thread(
|
|
304
|
+
export_session_bundle,
|
|
305
|
+
source_root,
|
|
306
|
+
destination,
|
|
307
|
+
name=f"{contract.agent}-environment",
|
|
308
|
+
)
|
|
309
|
+
except BundleError as failed:
|
|
310
|
+
print(f"Cannot seal the environment bundle: {failed}", file=sys.stderr)
|
|
311
|
+
return 1
|
|
312
|
+
print(f"\nworld: {destination}")
|
|
313
|
+
print(f"bundle: {bundle_path} ({bundle.digest})")
|
|
314
|
+
print(f"spent: ${stage.spent_usd:.4f}")
|
|
315
|
+
return 0
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
async def _environment(args: argparse.Namespace) -> int:
|
|
319
|
+
"""Provision or tear down the runtime shipped by the agent repository."""
|
|
320
|
+
from .provision import (
|
|
321
|
+
ProvisionedEnvironment,
|
|
322
|
+
ProvisionError,
|
|
323
|
+
provision,
|
|
324
|
+
reset,
|
|
325
|
+
stop,
|
|
326
|
+
)
|
|
327
|
+
|
|
328
|
+
destination = Path(args.out)
|
|
329
|
+
try:
|
|
330
|
+
if args.action == "down":
|
|
331
|
+
if not stop(destination):
|
|
332
|
+
print(f"No environment recorded at {destination}.", file=sys.stderr)
|
|
333
|
+
return 1
|
|
334
|
+
print(f"environment stopped: {destination}")
|
|
335
|
+
return 0
|
|
336
|
+
if args.action == "status":
|
|
337
|
+
environment = ProvisionedEnvironment.load(destination)
|
|
338
|
+
if environment is None:
|
|
339
|
+
print(f"No environment recorded at {destination}.", file=sys.stderr)
|
|
340
|
+
return 1
|
|
341
|
+
elif args.action == "reset":
|
|
342
|
+
environment = reset(destination)
|
|
343
|
+
else:
|
|
344
|
+
source_path = args.path
|
|
345
|
+
bundle_value = str(getattr(args, "bundle", "") or "")
|
|
346
|
+
if bundle_value:
|
|
347
|
+
from .bundle import BundleError, load_bundle
|
|
348
|
+
from .environment_plan import (
|
|
349
|
+
ENVIRONMENT_PLAN_FILE,
|
|
350
|
+
EnvironmentPlanError,
|
|
351
|
+
load_environment_plan,
|
|
352
|
+
)
|
|
353
|
+
|
|
354
|
+
bundle_root = Path(bundle_value).expanduser().resolve()
|
|
355
|
+
try:
|
|
356
|
+
bundle = load_bundle(bundle_root)
|
|
357
|
+
# New bundles carry the canonical decision record. Older sealed bundles
|
|
358
|
+
# remain rerunnable, but still receive full content verification.
|
|
359
|
+
if (bundle_root / ENVIRONMENT_PLAN_FILE).is_file():
|
|
360
|
+
load_environment_plan(bundle_root, bundle=bundle)
|
|
361
|
+
except (BundleError, EnvironmentPlanError) as failed:
|
|
362
|
+
print(f"Environment bundle failed: {failed}", file=sys.stderr)
|
|
363
|
+
return 1
|
|
364
|
+
bundled_source = bundle_root / "services" / "source"
|
|
365
|
+
if not bundled_source.is_dir():
|
|
366
|
+
print(
|
|
367
|
+
f"Environment bundle has no source snapshot: {bundled_source}",
|
|
368
|
+
file=sys.stderr,
|
|
369
|
+
)
|
|
370
|
+
return 1
|
|
371
|
+
source_path = str(bundled_source)
|
|
372
|
+
# Resuming a saved environment must use the same repository/runtime decision as the
|
|
373
|
+
# autonomous and hosted paths. In particular, a submitted Compose runtime may name
|
|
374
|
+
# a repository-local env file that is intentionally replaced by job-scoped values.
|
|
375
|
+
# Without the saved contract this command can incorrectly fall back to a Dockerfile
|
|
376
|
+
# and report that a previously valid Compose environment cannot be started.
|
|
377
|
+
environment = provision(source_path, destination, load(destination))
|
|
378
|
+
except ProvisionError as failed:
|
|
379
|
+
print(f"Environment failed: {failed}", file=sys.stderr)
|
|
380
|
+
return 1
|
|
381
|
+
|
|
382
|
+
print(f"environment: {environment.project}")
|
|
383
|
+
print(f"services: {', '.join(environment.services)}")
|
|
384
|
+
print(f"ready in: {environment.provision_seconds:.3f}s")
|
|
385
|
+
for name, value in sorted(environment.overrides.items()):
|
|
386
|
+
print(f"set: {name}={value}")
|
|
387
|
+
return 0
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
async def _scenarios(args: argparse.Namespace) -> int:
|
|
391
|
+
destination = Path(args.out) if args.out else artifact_dir(args.name)
|
|
392
|
+
contract = load(destination)
|
|
393
|
+
if contract is None:
|
|
394
|
+
print(f"No contract at {destination}. Run `understand` first.", file=sys.stderr)
|
|
395
|
+
return 1
|
|
396
|
+
if not world_saved(destination):
|
|
397
|
+
print(f"No world at {destination}. Run `build` first.", file=sys.stderr)
|
|
398
|
+
return 1
|
|
399
|
+
|
|
400
|
+
# With a suite already written, the target is what is there. Somebody who comes back to
|
|
401
|
+
# change one scenario is not asking for a different number of them.
|
|
402
|
+
existing = len(load_written(destination))
|
|
403
|
+
wanted = args.count or existing or 10
|
|
404
|
+
|
|
405
|
+
print(
|
|
406
|
+
f"agent: {contract.agent} "
|
|
407
|
+
+ (f"({existing} scenarios, loaded)" if existing else f"(writing {wanted})")
|
|
408
|
+
)
|
|
409
|
+
print(f"model: {chosen_model()}")
|
|
410
|
+
print(f"out: {destination}\n")
|
|
411
|
+
|
|
412
|
+
stage, _ = scenario_stage(
|
|
413
|
+
contract,
|
|
414
|
+
out=destination,
|
|
415
|
+
wanted=wanted,
|
|
416
|
+
ask=permission_gate(_ask_operator) if args.interactive else None,
|
|
417
|
+
)
|
|
418
|
+
await _converse(
|
|
419
|
+
stage,
|
|
420
|
+
scenario_opening(contract, wanted, existing) + _guidance(args),
|
|
421
|
+
interactive=args.interactive,
|
|
422
|
+
until=lambda: bool(load_written(destination)),
|
|
423
|
+
nudge=(
|
|
424
|
+
"Nothing was saved: you finished without calling save_scenarios. Submit anything "
|
|
425
|
+
"still unsubmitted, then call save_scenarios."
|
|
426
|
+
),
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
written = load_written(destination)
|
|
430
|
+
if not written:
|
|
431
|
+
print("\nNo scenarios were saved.", file=sys.stderr)
|
|
432
|
+
return 1
|
|
433
|
+
print(f"\nscenarios: {len(written)} in {destination / 'scenarios.json'}")
|
|
434
|
+
print(f"spent: ${stage.spent_usd:.4f}")
|
|
435
|
+
return 0
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
async def _live(args: argparse.Namespace) -> int:
|
|
439
|
+
"""The run stage as a conversation: it decides what to run and reads what came back."""
|
|
440
|
+
from .run.stage import load as load_results
|
|
441
|
+
from .run.stage import open_stage as run_stage
|
|
442
|
+
from .run.stage import opening as run_opening
|
|
443
|
+
|
|
444
|
+
destination = Path(args.out) if args.out else artifact_dir(args.name)
|
|
445
|
+
contract = load(destination)
|
|
446
|
+
written = load_written(destination)
|
|
447
|
+
if contract is None or not written:
|
|
448
|
+
print(
|
|
449
|
+
f"Need a contract and scenarios at {destination}. Run `understand`, `build` and "
|
|
450
|
+
"`scenarios` first.",
|
|
451
|
+
file=sys.stderr,
|
|
452
|
+
)
|
|
453
|
+
return 1
|
|
454
|
+
source_root = _source_root(destination)
|
|
455
|
+
if source_root:
|
|
456
|
+
_load_connection_env(Path(source_root))
|
|
457
|
+
|
|
458
|
+
print(f"agent: {contract.agent} ({len(written)} scenarios)")
|
|
459
|
+
print(f"model: {chosen_model()}")
|
|
460
|
+
print(f"out: {destination}\n")
|
|
461
|
+
|
|
462
|
+
stage, _ = run_stage(
|
|
463
|
+
contract,
|
|
464
|
+
out=destination,
|
|
465
|
+
ask=permission_gate(_ask_operator) if args.interactive else None,
|
|
466
|
+
)
|
|
467
|
+
await _converse(
|
|
468
|
+
stage, run_opening(contract, destination), interactive=args.interactive
|
|
469
|
+
)
|
|
470
|
+
|
|
471
|
+
results = load_results(destination)
|
|
472
|
+
passed = sum(1 for record in results if record["passed"])
|
|
473
|
+
print(f"\nruns: {passed} of {len(results)} passed, in {destination / 'runs.json'}")
|
|
474
|
+
print(f"spent: ${stage.spent_usd:.4f}")
|
|
475
|
+
return 0
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
async def _run(args: argparse.Namespace) -> int:
|
|
479
|
+
from .run import run_suite
|
|
480
|
+
from .run.grade import summarise
|
|
481
|
+
from .world.snapshot import require_source_implementation
|
|
482
|
+
|
|
483
|
+
destination = Path(args.out) if args.out else artifact_dir(args.name)
|
|
484
|
+
contract = load(destination)
|
|
485
|
+
written = load_written(destination)
|
|
486
|
+
if contract is None or not written:
|
|
487
|
+
print(
|
|
488
|
+
f"Need a contract and scenarios at {destination}. Run `understand`, `build` "
|
|
489
|
+
"and `scenarios` first.",
|
|
490
|
+
file=sys.stderr,
|
|
491
|
+
)
|
|
492
|
+
return 1
|
|
493
|
+
try:
|
|
494
|
+
require_source_implementation(destination)
|
|
495
|
+
except (FileNotFoundError, RuntimeError) as failed:
|
|
496
|
+
print(str(failed), file=sys.stderr)
|
|
497
|
+
return 1
|
|
498
|
+
|
|
499
|
+
chosen = [s for s in written if s.name in args.only] if args.only else written
|
|
500
|
+
if not chosen:
|
|
501
|
+
print(f"No scenario matching {args.only}.", file=sys.stderr)
|
|
502
|
+
return 1
|
|
503
|
+
|
|
504
|
+
print(f"agent: {contract.agent} ({len(chosen)} scenarios, target {args.target})")
|
|
505
|
+
print(f"model: {chosen_model()}")
|
|
506
|
+
print(f"out: {destination}\n")
|
|
507
|
+
|
|
508
|
+
def overheard(exchange: Any) -> None:
|
|
509
|
+
if args.quiet:
|
|
510
|
+
return
|
|
511
|
+
print(f" {exchange.speaker:8} {exchange.text}", flush=True)
|
|
512
|
+
|
|
513
|
+
def show(result: Any) -> None:
|
|
514
|
+
# Just the verdict as it lands. The detail is in the summary at the end, and printing
|
|
515
|
+
# it in both places means every failure is read twice.
|
|
516
|
+
print(result.line(), flush=True)
|
|
517
|
+
|
|
518
|
+
results = await run_suite(
|
|
519
|
+
chosen,
|
|
520
|
+
contract,
|
|
521
|
+
destination,
|
|
522
|
+
target=args.target,
|
|
523
|
+
model=args.model,
|
|
524
|
+
on_result=show,
|
|
525
|
+
on_exchange=overheard,
|
|
526
|
+
)
|
|
527
|
+
print("\n" + summarise(results))
|
|
528
|
+
print(f"\nspent: ${sum(result.spent_usd for result in results):.4f}")
|
|
529
|
+
return 0 if all(result.passed for result in results) else 2
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
async def _simulate(args: argparse.Namespace) -> int:
|
|
533
|
+
"""Run the suite through the modality/runtime inferred from the saved contract."""
|
|
534
|
+
from . import platform
|
|
535
|
+
from .run.simulation import simulate
|
|
536
|
+
from .world.snapshot import require_source_implementation
|
|
537
|
+
|
|
538
|
+
destination = Path(args.out) if args.out else artifact_dir(args.name)
|
|
539
|
+
contract = load(destination)
|
|
540
|
+
written = load_written(destination)
|
|
541
|
+
if contract is None or not written:
|
|
542
|
+
print(
|
|
543
|
+
f"Need a contract and scenarios at {destination}.",
|
|
544
|
+
file=sys.stderr,
|
|
545
|
+
)
|
|
546
|
+
return 1
|
|
547
|
+
try:
|
|
548
|
+
require_source_implementation(destination)
|
|
549
|
+
except (FileNotFoundError, RuntimeError) as failed:
|
|
550
|
+
print(str(failed), file=sys.stderr)
|
|
551
|
+
return 1
|
|
552
|
+
chosen = [s for s in written if s.name in args.only] if args.only else written
|
|
553
|
+
if not chosen:
|
|
554
|
+
print(f"No scenario matching {args.only}.", file=sys.stderr)
|
|
555
|
+
return 1
|
|
556
|
+
|
|
557
|
+
# A resumed simulation is a first-class execution path, not merely an internal stage of
|
|
558
|
+
# ``auto``. Load the submitted repository's connection settings here as well so restarting a
|
|
559
|
+
# completed build does not require the operator to rediscover and export its LiveKit/model
|
|
560
|
+
# credentials by hand. Existing worker/host values continue to win in _load_connection_env.
|
|
561
|
+
source_root = _source_root(destination)
|
|
562
|
+
if source_root:
|
|
563
|
+
_load_connection_env(Path(source_root))
|
|
564
|
+
|
|
565
|
+
reported = None
|
|
566
|
+
call_ids: dict[str, str] = {}
|
|
567
|
+
blocked = platform.configured()
|
|
568
|
+
if not blocked:
|
|
569
|
+
try:
|
|
570
|
+
reported, allocated = platform.begin(
|
|
571
|
+
chosen,
|
|
572
|
+
name=platform.display_run_name(args.name),
|
|
573
|
+
run_test_id=platform.remembered(destination),
|
|
574
|
+
modality=contract.modality or "text",
|
|
575
|
+
)
|
|
576
|
+
call_ids = {
|
|
577
|
+
scenario.name: call_id
|
|
578
|
+
for scenario, call_id in zip(chosen, allocated, strict=False)
|
|
579
|
+
}
|
|
580
|
+
# Persist the destination as soon as the platform execution exists. The list view
|
|
581
|
+
# can now show an in-progress run, and a process restart reuses the same RunTest.
|
|
582
|
+
platform.remember(destination, reported)
|
|
583
|
+
print(f"platform run: {reported.url}", flush=True)
|
|
584
|
+
except platform.PlatformError as failed:
|
|
585
|
+
print(f"platform reporting could not start: {failed}", file=sys.stderr)
|
|
586
|
+
else:
|
|
587
|
+
print(f"not reported to the platform: {blocked}", flush=True)
|
|
588
|
+
|
|
589
|
+
def show(result: Any) -> None:
|
|
590
|
+
print(result.line(), flush=True)
|
|
591
|
+
if reported is None:
|
|
592
|
+
return
|
|
593
|
+
call_id = call_ids.get(result.scenario)
|
|
594
|
+
if not call_id:
|
|
595
|
+
reported.problems.append(
|
|
596
|
+
f"the platform allocated no call for {result.scenario}"
|
|
597
|
+
)
|
|
598
|
+
return
|
|
599
|
+
platform.send_result(reported, call_id, result)
|
|
600
|
+
|
|
601
|
+
def show_started(scenario: Any) -> None:
|
|
602
|
+
if reported is None:
|
|
603
|
+
return
|
|
604
|
+
platform.mark_ongoing(reported, call_ids.get(scenario.name, ""))
|
|
605
|
+
|
|
606
|
+
summary = await simulate(
|
|
607
|
+
chosen,
|
|
608
|
+
contract,
|
|
609
|
+
destination,
|
|
610
|
+
destination=destination,
|
|
611
|
+
model=args.model,
|
|
612
|
+
on_case_start=show_started,
|
|
613
|
+
on_case_done=show,
|
|
614
|
+
)
|
|
615
|
+
print(
|
|
616
|
+
f"\n{summary['passed']}/{summary['scenarios']} scenarios passed "
|
|
617
|
+
f"in {summary['seconds']}s"
|
|
618
|
+
)
|
|
619
|
+
print(f"run: {destination / 'runs' / summary['run_id']}")
|
|
620
|
+
if reported is not None:
|
|
621
|
+
for problem in reported.problems:
|
|
622
|
+
print(f"platform reporting problem: {problem}", file=sys.stderr)
|
|
623
|
+
print(f"reported to the platform: {reported.url}")
|
|
624
|
+
# Exit 1 means the environment/call lane could not execute at least one scenario. Exit 2
|
|
625
|
+
# means every scenario ran and the submitted agent failed one or more checks. Hosted
|
|
626
|
+
# execution retries/classifies the former and preserves the latter as valid RL evidence.
|
|
627
|
+
if summary.get("unrunnable"):
|
|
628
|
+
return 1
|
|
629
|
+
return 0 if summary["passed"] == summary["scenarios"] else 2
|
|
630
|
+
|
|
631
|
+
|
|
632
|
+
def _load_connection_env(source: Path) -> list[str]:
|
|
633
|
+
"""Fill missing connection variables from the submitted repository.
|
|
634
|
+
|
|
635
|
+
A dotenv file is data, not a shell program. Sourcing a customer's file executes command
|
|
636
|
+
substitutions and also breaks on perfectly valid unquoted values containing spaces. Values
|
|
637
|
+
already supplied by the workspace win: in particular, a worker's container-only credential
|
|
638
|
+
path must never replace the host's model-provider credential path.
|
|
639
|
+
"""
|
|
640
|
+
loaded: list[str] = []
|
|
641
|
+
for candidate in (source / ".env.local", source / ".env"):
|
|
642
|
+
if not candidate.is_file():
|
|
643
|
+
continue
|
|
644
|
+
for raw in candidate.read_text(encoding="utf-8").splitlines():
|
|
645
|
+
line = raw.strip()
|
|
646
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
647
|
+
continue
|
|
648
|
+
name, value = line.split("=", 1)
|
|
649
|
+
name = name.removeprefix("export ").strip()
|
|
650
|
+
if not name.replace("_", "").isalnum() or not name[0].isalpha():
|
|
651
|
+
continue
|
|
652
|
+
value = value.strip()
|
|
653
|
+
if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'":
|
|
654
|
+
value = value[1:-1]
|
|
655
|
+
if name not in os.environ:
|
|
656
|
+
os.environ[name] = value
|
|
657
|
+
loaded.append(name)
|
|
658
|
+
return loaded
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
def _new_adjustments(
|
|
662
|
+
path: Path | None, cursor: int
|
|
663
|
+
) -> tuple[list[dict[str, Any]], int]:
|
|
664
|
+
if path is None or not path.is_file():
|
|
665
|
+
return [], cursor
|
|
666
|
+
records: list[dict[str, Any]] = []
|
|
667
|
+
lines = path.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
668
|
+
for line in lines[cursor:]:
|
|
669
|
+
try:
|
|
670
|
+
value = json.loads(line)
|
|
671
|
+
except ValueError:
|
|
672
|
+
continue
|
|
673
|
+
if isinstance(value, dict):
|
|
674
|
+
records.append(value)
|
|
675
|
+
return records, len(lines)
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def _write_adjustment_status(
|
|
679
|
+
inbox: Path | None,
|
|
680
|
+
adjustment_id: str,
|
|
681
|
+
status: str,
|
|
682
|
+
*,
|
|
683
|
+
applied_stage: str,
|
|
684
|
+
) -> None:
|
|
685
|
+
if inbox is None:
|
|
686
|
+
return
|
|
687
|
+
status_path = inbox.with_name("adjustment-status.jsonl")
|
|
688
|
+
record = {
|
|
689
|
+
"adjustment_id": adjustment_id,
|
|
690
|
+
"status": status,
|
|
691
|
+
"applied_stage": applied_stage,
|
|
692
|
+
"updated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
693
|
+
}
|
|
694
|
+
with status_path.open("a", encoding="utf-8") as stream:
|
|
695
|
+
stream.write(json.dumps(record, separators=(",", ":")) + "\n")
|
|
696
|
+
stream.flush()
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
async def _auto(args: argparse.Namespace) -> int:
|
|
700
|
+
"""Take one source connection through every stage without operator messages.
|
|
701
|
+
|
|
702
|
+
This is the product acceptance path. The individual stage commands remain useful while
|
|
703
|
+
developing the harness, but a submitted agent must not depend on somebody knowing their
|
|
704
|
+
order, nudging a model, or repairing an artifact between stages.
|
|
705
|
+
"""
|
|
706
|
+
source = Path(args.path).expanduser().resolve()
|
|
707
|
+
loaded_connection = _load_connection_env(source)
|
|
708
|
+
name = (args.name or source.name).strip()
|
|
709
|
+
destination = (
|
|
710
|
+
Path(args.out).expanduser().resolve()
|
|
711
|
+
if args.out
|
|
712
|
+
else artifact_dir(new_id(name))
|
|
713
|
+
)
|
|
714
|
+
destination.mkdir(parents=True, exist_ok=False)
|
|
715
|
+
from fi.simulate.runtime.events import CanonicalEvent
|
|
716
|
+
|
|
717
|
+
from .events import BufferedEventSink, EventOutbox
|
|
718
|
+
from .job import (
|
|
719
|
+
AgentConnection,
|
|
720
|
+
ExecutionMode,
|
|
721
|
+
HarnessJob,
|
|
722
|
+
RepositorySource,
|
|
723
|
+
SourceKind,
|
|
724
|
+
)
|
|
725
|
+
|
|
726
|
+
job = getattr(args, "job", None) or HarnessJob(
|
|
727
|
+
job_id=destination.name,
|
|
728
|
+
run_id=destination.name,
|
|
729
|
+
execution=ExecutionMode.LOCAL,
|
|
730
|
+
source=RepositorySource(
|
|
731
|
+
kind=SourceKind.LOCAL_REPOSITORY, local_path=str(source)
|
|
732
|
+
),
|
|
733
|
+
agent=AgentConnection(connector="auto"),
|
|
734
|
+
scenario_count=args.count,
|
|
735
|
+
metadata={"agent_name": name, "source_kind": args.kind},
|
|
736
|
+
)
|
|
737
|
+
(destination / "job.json").write_text(
|
|
738
|
+
job.model_dump_json(indent=2) + "\n", encoding="utf-8"
|
|
739
|
+
)
|
|
740
|
+
# Beside the authoring output, so the platform reads the running total while the sandbox lives.
|
|
741
|
+
spend.journal_to(destination / "cost.json")
|
|
742
|
+
events = BufferedEventSink(EventOutbox(destination.parent, destination.name))
|
|
743
|
+
event_sequence = 0
|
|
744
|
+
|
|
745
|
+
def emit(event_type: str, stage: str, **payload: Any) -> None:
|
|
746
|
+
nonlocal event_sequence
|
|
747
|
+
observability.stage_event(event_type, stage, payload)
|
|
748
|
+
events.write(
|
|
749
|
+
CanonicalEvent.create(
|
|
750
|
+
run_id=job.run_id,
|
|
751
|
+
test_case_id="harness",
|
|
752
|
+
event_type=event_type,
|
|
753
|
+
source="fi.alk.harness",
|
|
754
|
+
sequence=event_sequence,
|
|
755
|
+
payload={"stage": stage, **payload},
|
|
756
|
+
)
|
|
757
|
+
)
|
|
758
|
+
event_sequence += 1
|
|
759
|
+
|
|
760
|
+
now = time.time()
|
|
761
|
+
save_session(
|
|
762
|
+
Session(
|
|
763
|
+
id=destination.name,
|
|
764
|
+
path=destination,
|
|
765
|
+
agent=name,
|
|
766
|
+
source=str(source),
|
|
767
|
+
kind=args.kind,
|
|
768
|
+
created=now,
|
|
769
|
+
updated=now,
|
|
770
|
+
stage="understand",
|
|
771
|
+
title=name,
|
|
772
|
+
)
|
|
773
|
+
)
|
|
774
|
+
|
|
775
|
+
print("automatic acceptance run")
|
|
776
|
+
print(f"agent: {name}")
|
|
777
|
+
print(f"source: {source}")
|
|
778
|
+
print(f"out: {destination}")
|
|
779
|
+
print("operator input: disabled\n")
|
|
780
|
+
if loaded_connection:
|
|
781
|
+
print(
|
|
782
|
+
"connection: loaded missing variables from the submitted repository "
|
|
783
|
+
f"({len(loaded_connection)} names; values hidden)\n"
|
|
784
|
+
)
|
|
785
|
+
|
|
786
|
+
authoring_only = bool(getattr(args, "authoring_only", False))
|
|
787
|
+
stages = [
|
|
788
|
+
(
|
|
789
|
+
"understand",
|
|
790
|
+
_understand,
|
|
791
|
+
argparse.Namespace(
|
|
792
|
+
name=name,
|
|
793
|
+
path=str(source),
|
|
794
|
+
kind=args.kind,
|
|
795
|
+
out=str(destination),
|
|
796
|
+
interactive=False,
|
|
797
|
+
model=args.model,
|
|
798
|
+
guidance=[],
|
|
799
|
+
job=job,
|
|
800
|
+
provider_profile=getattr(args, "provider_profile", None),
|
|
801
|
+
),
|
|
802
|
+
),
|
|
803
|
+
(
|
|
804
|
+
"environment",
|
|
805
|
+
_build,
|
|
806
|
+
argparse.Namespace(
|
|
807
|
+
name=name,
|
|
808
|
+
path=str(source),
|
|
809
|
+
out=str(destination),
|
|
810
|
+
interactive=False,
|
|
811
|
+
guidance=[],
|
|
812
|
+
# Hosted V2 authoring resolves and provisions the submitted runtime inside the
|
|
813
|
+
# Daytona guest. The authoring worker still creates the same logical world and
|
|
814
|
+
# scenarios, but must not start customer Compose/Docker resources on the control
|
|
815
|
+
# plane worker merely to describe them.
|
|
816
|
+
skip_source_provision=authoring_only,
|
|
817
|
+
# A source-free provider connection deliberately keeps the provider's deployed
|
|
818
|
+
# HTTP tools in place. Their real execution is observed during calls; no local
|
|
819
|
+
# source entrypoint exists or is required.
|
|
820
|
+
external_runtime=authoring_only and args.kind == "provider",
|
|
821
|
+
),
|
|
822
|
+
),
|
|
823
|
+
(
|
|
824
|
+
"scenarios",
|
|
825
|
+
_scenarios,
|
|
826
|
+
argparse.Namespace(
|
|
827
|
+
name=name,
|
|
828
|
+
out=str(destination),
|
|
829
|
+
count=args.count,
|
|
830
|
+
interactive=False,
|
|
831
|
+
guidance=[],
|
|
832
|
+
),
|
|
833
|
+
),
|
|
834
|
+
]
|
|
835
|
+
if not authoring_only:
|
|
836
|
+
stages.append(
|
|
837
|
+
(
|
|
838
|
+
"calls",
|
|
839
|
+
_simulate,
|
|
840
|
+
argparse.Namespace(
|
|
841
|
+
name=name,
|
|
842
|
+
out=str(destination),
|
|
843
|
+
only=None,
|
|
844
|
+
model=args.run_model,
|
|
845
|
+
),
|
|
846
|
+
)
|
|
847
|
+
)
|
|
848
|
+
from .provision import ProvisionError, stop
|
|
849
|
+
|
|
850
|
+
cleanup_failed: ProvisionError | None = None
|
|
851
|
+
try:
|
|
852
|
+
stage_index = 0
|
|
853
|
+
adjustment_cursor = 0
|
|
854
|
+
applying: dict[str, list[str]] = {label: [] for label, *_ in stages}
|
|
855
|
+
scenario_adjustment_guidance: dict[str, str] = {}
|
|
856
|
+
adjustments_path = (
|
|
857
|
+
Path(args.adjustments_path)
|
|
858
|
+
if getattr(args, "adjustments_path", None)
|
|
859
|
+
else None
|
|
860
|
+
)
|
|
861
|
+
while stage_index < len(stages):
|
|
862
|
+
label, operation, stage_args = stages[stage_index]
|
|
863
|
+
print(f"\n=== {label} ===", flush=True)
|
|
864
|
+
emit("harness.stage.started", label)
|
|
865
|
+
status = await operation(stage_args)
|
|
866
|
+
# Hosted authoring is unattended, and a successful scenario-stage
|
|
867
|
+
# process is not sufficient evidence that it honoured the requested
|
|
868
|
+
# cardinality. Models can checkpoint a valid partial suite (for
|
|
869
|
+
# example 2/3); Bundle V2 must remain strict, so repair the producer
|
|
870
|
+
# output here while the authoring context is still available.
|
|
871
|
+
if label == "scenarios" and authoring_only and not status:
|
|
872
|
+
repair_attempt = 0
|
|
873
|
+
wanted = int(stage_args.count)
|
|
874
|
+
written_count = len(load_written(destination))
|
|
875
|
+
while written_count != wanted and repair_attempt < 2:
|
|
876
|
+
repair_attempt += 1
|
|
877
|
+
missing = wanted - written_count
|
|
878
|
+
# Count alone is the wrong instruction: asked only for a number, the
|
|
879
|
+
# stage pads with happy paths that satisfy cardinality and measure nothing.
|
|
880
|
+
stage_args.guidance = [
|
|
881
|
+
(
|
|
882
|
+
f"The hosted run requires exactly {wanted} scenarios, but "
|
|
883
|
+
f"only {written_count} are currently saved. "
|
|
884
|
+
+ (
|
|
885
|
+
f"Add exactly {missing} distinct validated scenario(s) and "
|
|
886
|
+
"call save_scenarios. Preserve all existing scenarios. Each "
|
|
887
|
+
"one must meet the same bar as the rest of the suite: a "
|
|
888
|
+
"different branch of the agent's behaviour from every "
|
|
889
|
+
"scenario already saved, several steps deep, and failing "
|
|
890
|
+
"when the agent does the wrong thing. Do not pad with "
|
|
891
|
+
"variations of a scenario that already exists, and do not "
|
|
892
|
+
"add a happy path that an existing scenario already covers."
|
|
893
|
+
if missing > 0
|
|
894
|
+
else f"Remove exactly {-missing} excess scenario(s), preserve "
|
|
895
|
+
"the strongest coverage, and call save_scenarios. Drop the "
|
|
896
|
+
"ones that duplicate a branch another scenario already "
|
|
897
|
+
"exercises, not the ones that are hardest to pass."
|
|
898
|
+
)
|
|
899
|
+
)
|
|
900
|
+
]
|
|
901
|
+
emit(
|
|
902
|
+
"harness.stage.repairing",
|
|
903
|
+
label,
|
|
904
|
+
attempt=repair_attempt,
|
|
905
|
+
expected_scenarios=wanted,
|
|
906
|
+
written_scenarios=written_count,
|
|
907
|
+
)
|
|
908
|
+
status = await operation(stage_args)
|
|
909
|
+
if status:
|
|
910
|
+
break
|
|
911
|
+
written_count = len(load_written(destination))
|
|
912
|
+
if not status and written_count != wanted:
|
|
913
|
+
emit(
|
|
914
|
+
"harness.stage.failed",
|
|
915
|
+
label,
|
|
916
|
+
status=1,
|
|
917
|
+
code="scenario_count_mismatch",
|
|
918
|
+
expected_scenarios=wanted,
|
|
919
|
+
written_scenarios=written_count,
|
|
920
|
+
)
|
|
921
|
+
print(
|
|
922
|
+
"\nautomatic run stopped: scenario generation saved "
|
|
923
|
+
f"{written_count}/{wanted} requested scenarios after "
|
|
924
|
+
f"{repair_attempt} repair attempts",
|
|
925
|
+
file=sys.stderr,
|
|
926
|
+
)
|
|
927
|
+
status = 1
|
|
928
|
+
if (
|
|
929
|
+
label == "scenarios"
|
|
930
|
+
and authoring_only
|
|
931
|
+
and not status
|
|
932
|
+
and applying[label]
|
|
933
|
+
):
|
|
934
|
+
missing = _missing_scenario_adjustments(
|
|
935
|
+
load_written(destination), applying[label]
|
|
936
|
+
)
|
|
937
|
+
repair_attempt = 0
|
|
938
|
+
while missing and repair_attempt < 2:
|
|
939
|
+
repair_attempt += 1
|
|
940
|
+
stage_args.guidance = [
|
|
941
|
+
scenario_adjustment_guidance[adjustment_id]
|
|
942
|
+
+ "\nNo saved scenario currently carries this marker. Correct the "
|
|
943
|
+
"suite, preserve unaffected scenarios, and call save_scenarios again."
|
|
944
|
+
for adjustment_id in missing
|
|
945
|
+
]
|
|
946
|
+
emit(
|
|
947
|
+
"harness.stage.repairing",
|
|
948
|
+
label,
|
|
949
|
+
attempt=repair_attempt,
|
|
950
|
+
missing_adjustment_ids=missing,
|
|
951
|
+
)
|
|
952
|
+
status = await operation(stage_args)
|
|
953
|
+
if status:
|
|
954
|
+
break
|
|
955
|
+
missing = _missing_scenario_adjustments(
|
|
956
|
+
load_written(destination), applying[label]
|
|
957
|
+
)
|
|
958
|
+
if not status and missing:
|
|
959
|
+
emit(
|
|
960
|
+
"harness.stage.failed",
|
|
961
|
+
label,
|
|
962
|
+
status=1,
|
|
963
|
+
code="scenario_adjustment_not_reflected",
|
|
964
|
+
missing_adjustment_ids=missing,
|
|
965
|
+
)
|
|
966
|
+
print(
|
|
967
|
+
"\nautomatic run stopped: saved scenarios did not reflect "
|
|
968
|
+
f"adjustments {', '.join(missing)}",
|
|
969
|
+
file=sys.stderr,
|
|
970
|
+
)
|
|
971
|
+
status = 1
|
|
972
|
+
# A completed call suite returns 2 when the submitted agent fails one or more checks.
|
|
973
|
+
# That is a valid RL result. Earlier stages returning non-zero are harness failures.
|
|
974
|
+
if status and label != "calls":
|
|
975
|
+
emit("harness.stage.failed", label, status=status)
|
|
976
|
+
print(f"\nautomatic run stopped: {label} failed", file=sys.stderr)
|
|
977
|
+
return status
|
|
978
|
+
if label == "calls" and status not in (0, 2):
|
|
979
|
+
emit("harness.stage.failed", label, status=status)
|
|
980
|
+
return status
|
|
981
|
+
emit("harness.stage.completed", label, status=status)
|
|
982
|
+
for adjustment_id in applying[label]:
|
|
983
|
+
_write_adjustment_status(
|
|
984
|
+
adjustments_path,
|
|
985
|
+
adjustment_id,
|
|
986
|
+
"applied",
|
|
987
|
+
applied_stage=label,
|
|
988
|
+
)
|
|
989
|
+
emit(
|
|
990
|
+
"harness.adjustment.applied",
|
|
991
|
+
label,
|
|
992
|
+
adjustment_id=adjustment_id,
|
|
993
|
+
)
|
|
994
|
+
applying[label] = []
|
|
995
|
+
if hasattr(stage_args, "guidance"):
|
|
996
|
+
stage_args.guidance = []
|
|
997
|
+
|
|
998
|
+
incoming, adjustment_cursor = _new_adjustments(
|
|
999
|
+
adjustments_path, adjustment_cursor
|
|
1000
|
+
)
|
|
1001
|
+
rewind_to: int | None = None
|
|
1002
|
+
for adjustment in incoming:
|
|
1003
|
+
target = str(adjustment.get("target_stage") or label)
|
|
1004
|
+
target_index = next(
|
|
1005
|
+
(
|
|
1006
|
+
index
|
|
1007
|
+
for index, (stage_name, *_rest) in enumerate(stages)
|
|
1008
|
+
if stage_name == target
|
|
1009
|
+
),
|
|
1010
|
+
min(stage_index, 2),
|
|
1011
|
+
)
|
|
1012
|
+
target_args = stages[target_index][2]
|
|
1013
|
+
instruction = str(adjustment.get("instruction") or "")
|
|
1014
|
+
adjustment_id = str(adjustment.get("adjustment_id") or "")
|
|
1015
|
+
if target == "scenarios" and adjustment_id:
|
|
1016
|
+
instruction = _scenario_adjustment_requirement(
|
|
1017
|
+
adjustment_id, instruction
|
|
1018
|
+
)
|
|
1019
|
+
scenario_adjustment_guidance[adjustment_id] = instruction
|
|
1020
|
+
target_args.guidance = [
|
|
1021
|
+
*getattr(target_args, "guidance", []),
|
|
1022
|
+
instruction,
|
|
1023
|
+
]
|
|
1024
|
+
delta = adjustment.get("scenario_delta")
|
|
1025
|
+
if target == "scenarios" and isinstance(delta, int) and delta > 0:
|
|
1026
|
+
existing_count = len(load_written(destination))
|
|
1027
|
+
target_args.count = max(target_args.count, existing_count) + delta
|
|
1028
|
+
if adjustment_id:
|
|
1029
|
+
applying[target].append(adjustment_id)
|
|
1030
|
+
_write_adjustment_status(
|
|
1031
|
+
adjustments_path,
|
|
1032
|
+
adjustment_id,
|
|
1033
|
+
"applying",
|
|
1034
|
+
applied_stage=target,
|
|
1035
|
+
)
|
|
1036
|
+
emit(
|
|
1037
|
+
"harness.adjustment.applying",
|
|
1038
|
+
target,
|
|
1039
|
+
adjustment_id=adjustment_id,
|
|
1040
|
+
)
|
|
1041
|
+
if target_index <= stage_index:
|
|
1042
|
+
rewind_to = (
|
|
1043
|
+
target_index
|
|
1044
|
+
if rewind_to is None
|
|
1045
|
+
else min(rewind_to, target_index)
|
|
1046
|
+
)
|
|
1047
|
+
if rewind_to is not None:
|
|
1048
|
+
emit(
|
|
1049
|
+
"harness.pipeline.rewound",
|
|
1050
|
+
stages[rewind_to][0],
|
|
1051
|
+
from_stage=label,
|
|
1052
|
+
)
|
|
1053
|
+
stage_index = rewind_to
|
|
1054
|
+
else:
|
|
1055
|
+
stage_index += 1
|
|
1056
|
+
finally:
|
|
1057
|
+
# The source environment exists only for this run. This boundary covers normal stage
|
|
1058
|
+
# failures and exceptions raised by world/scenario construction. Cleanup happens before
|
|
1059
|
+
# sealing so environment.json's terminal state is part of the immutable manifest.
|
|
1060
|
+
emit("harness.stage.started", "cleaning_up")
|
|
1061
|
+
try:
|
|
1062
|
+
await asyncio.to_thread(stop, destination)
|
|
1063
|
+
except ProvisionError as exc:
|
|
1064
|
+
cleanup_failed = exc
|
|
1065
|
+
emit(
|
|
1066
|
+
"harness.stage.failed",
|
|
1067
|
+
"cleaning_up",
|
|
1068
|
+
status=1,
|
|
1069
|
+
code="environment_cleanup_failed",
|
|
1070
|
+
detail=str(exc),
|
|
1071
|
+
)
|
|
1072
|
+
print(f"\nautomatic run cleanup failed: {exc}", file=sys.stderr)
|
|
1073
|
+
else:
|
|
1074
|
+
emit("harness.stage.completed", "cleaning_up", status=0)
|
|
1075
|
+
|
|
1076
|
+
if cleanup_failed is not None:
|
|
1077
|
+
return 1
|
|
1078
|
+
|
|
1079
|
+
if authoring_only:
|
|
1080
|
+
emit("harness.authoring.completed", "scenarios")
|
|
1081
|
+
print(f"\nautomatic authoring complete: {destination}")
|
|
1082
|
+
return 0
|
|
1083
|
+
|
|
1084
|
+
from .artifacts import ArtifactIntegrityError, seal_artifacts
|
|
1085
|
+
|
|
1086
|
+
emit("harness.stage.started", "uploading_artifacts")
|
|
1087
|
+
try:
|
|
1088
|
+
manifest = seal_artifacts(
|
|
1089
|
+
destination,
|
|
1090
|
+
run_id=job.run_id,
|
|
1091
|
+
max_bytes=job.artifacts.max_artifact_bytes,
|
|
1092
|
+
expected_scenarios=len(load_written(destination)) or job.scenario_count,
|
|
1093
|
+
)
|
|
1094
|
+
except ArtifactIntegrityError as exc:
|
|
1095
|
+
emit(
|
|
1096
|
+
"harness.stage.failed",
|
|
1097
|
+
"uploading_artifacts",
|
|
1098
|
+
status=1,
|
|
1099
|
+
code="artifact_integrity_failed",
|
|
1100
|
+
detail=str(exc),
|
|
1101
|
+
)
|
|
1102
|
+
print(
|
|
1103
|
+
f"\nautomatic run stopped: artifact integrity failed: {exc}",
|
|
1104
|
+
file=sys.stderr,
|
|
1105
|
+
)
|
|
1106
|
+
return 1
|
|
1107
|
+
emit(
|
|
1108
|
+
"harness.stage.completed",
|
|
1109
|
+
"uploading_artifacts",
|
|
1110
|
+
status=0,
|
|
1111
|
+
artifact_manifest_digest=manifest.digest,
|
|
1112
|
+
artifact_count=len(manifest.files),
|
|
1113
|
+
artifact_bytes=manifest.total_bytes,
|
|
1114
|
+
)
|
|
1115
|
+
emit("harness.run.completed", "completed")
|
|
1116
|
+
print(f"\nautomatic run complete: {destination}")
|
|
1117
|
+
return 0
|
|
1118
|
+
|
|
1119
|
+
|
|
1120
|
+
async def _chat(args: argparse.Namespace) -> int:
|
|
1121
|
+
"""One conversation for the whole thing: point at an agent and keep talking."""
|
|
1122
|
+
conversation = open_conversation(
|
|
1123
|
+
name=args.name or "",
|
|
1124
|
+
path=args.path or "",
|
|
1125
|
+
kind=args.kind,
|
|
1126
|
+
out=Path(args.out) if args.out else None,
|
|
1127
|
+
ask=permission_gate(_ask_operator),
|
|
1128
|
+
)
|
|
1129
|
+
print(f"model: {chosen_model()}")
|
|
1130
|
+
print(credentials_hint())
|
|
1131
|
+
print("\nSay what you want. Enter on its own moves to the next stage; 'q' ends.\n")
|
|
1132
|
+
|
|
1133
|
+
await conversation.start(on_event=_render)
|
|
1134
|
+
while True:
|
|
1135
|
+
try:
|
|
1136
|
+
said = await _prompt(f"\nyou ({conversation.stage_name}) ")
|
|
1137
|
+
except (EOFError, KeyboardInterrupt):
|
|
1138
|
+
break
|
|
1139
|
+
if said in {"q", "quit", "exit"}:
|
|
1140
|
+
break
|
|
1141
|
+
if not said:
|
|
1142
|
+
entered = await conversation.advance(on_event=_render)
|
|
1143
|
+
if entered is None:
|
|
1144
|
+
print(
|
|
1145
|
+
"\n [nothing to move on to yet; this stage has not produced its artifact]"
|
|
1146
|
+
)
|
|
1147
|
+
continue
|
|
1148
|
+
await conversation.say(said, on_event=_render)
|
|
1149
|
+
await conversation.close()
|
|
1150
|
+
print(f"\nspent: ${conversation.spent_usd:.4f}")
|
|
1151
|
+
return 0
|
|
1152
|
+
|
|
1153
|
+
|
|
1154
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
1155
|
+
parser = argparse.ArgumentParser(prog="agent-harness", description=__doc__)
|
|
1156
|
+
# Talking to it is the way in, so that is what happens when you just start it.
|
|
1157
|
+
sub = parser.add_subparsers(dest="stage", required=False)
|
|
1158
|
+
|
|
1159
|
+
understand = sub.add_parser(
|
|
1160
|
+
"understand", help="read an agent and produce its contract"
|
|
1161
|
+
)
|
|
1162
|
+
understand.add_argument("--name", required=True, help="what to call this agent")
|
|
1163
|
+
understand.add_argument("--path", required=True, help="where the agent is")
|
|
1164
|
+
understand.add_argument(
|
|
1165
|
+
"--kind", default="repo", choices=supported(), help="how the agent is supplied"
|
|
1166
|
+
)
|
|
1167
|
+
understand.add_argument("--out", default=None, help="artifact directory")
|
|
1168
|
+
understand.add_argument(
|
|
1169
|
+
"--once",
|
|
1170
|
+
dest="interactive",
|
|
1171
|
+
action="store_false",
|
|
1172
|
+
help="run unattended instead of staying open for corrections",
|
|
1173
|
+
)
|
|
1174
|
+
understand.add_argument("--model", default=None, help=argparse.SUPPRESS)
|
|
1175
|
+
understand.set_defaults(run=_understand, interactive=True)
|
|
1176
|
+
|
|
1177
|
+
world = sub.add_parser("build", help="build the world from an agent's contract")
|
|
1178
|
+
world.add_argument("--name", required=True, help="which agent")
|
|
1179
|
+
world.add_argument("--out", default=None, help="artifact directory")
|
|
1180
|
+
world.add_argument(
|
|
1181
|
+
"--path",
|
|
1182
|
+
default=None,
|
|
1183
|
+
help="agent source path (normally recovered from the session automatically)",
|
|
1184
|
+
)
|
|
1185
|
+
world.add_argument(
|
|
1186
|
+
"--once",
|
|
1187
|
+
dest="interactive",
|
|
1188
|
+
action="store_false",
|
|
1189
|
+
help="run unattended instead of staying open for corrections",
|
|
1190
|
+
)
|
|
1191
|
+
world.set_defaults(run=_build, interactive=True)
|
|
1192
|
+
|
|
1193
|
+
environment = sub.add_parser(
|
|
1194
|
+
"environment", help="start, inspect, or stop the runtime shipped by an agent"
|
|
1195
|
+
)
|
|
1196
|
+
environment.add_argument("action", choices=("up", "status", "reset", "down"))
|
|
1197
|
+
environment.add_argument(
|
|
1198
|
+
"--path", default="", help="agent repository (required for up)"
|
|
1199
|
+
)
|
|
1200
|
+
environment.add_argument(
|
|
1201
|
+
"--bundle",
|
|
1202
|
+
default="",
|
|
1203
|
+
help="sealed environment bundle to verify and restart (preferred for reruns)",
|
|
1204
|
+
)
|
|
1205
|
+
environment.add_argument("--out", required=True, help="session artifact directory")
|
|
1206
|
+
environment.set_defaults(run=_environment)
|
|
1207
|
+
|
|
1208
|
+
scenarios = sub.add_parser(
|
|
1209
|
+
"scenarios", help="write the scenarios to test the agent with"
|
|
1210
|
+
)
|
|
1211
|
+
scenarios.add_argument("--name", required=True, help="which agent")
|
|
1212
|
+
scenarios.add_argument("--out", default=None, help="artifact directory")
|
|
1213
|
+
scenarios.add_argument(
|
|
1214
|
+
"--count",
|
|
1215
|
+
type=int,
|
|
1216
|
+
default=None,
|
|
1217
|
+
help="how many scenarios to write (defaults to however many already exist)",
|
|
1218
|
+
)
|
|
1219
|
+
scenarios.add_argument(
|
|
1220
|
+
"--once",
|
|
1221
|
+
dest="interactive",
|
|
1222
|
+
action="store_false",
|
|
1223
|
+
help="run unattended instead of staying open for corrections",
|
|
1224
|
+
)
|
|
1225
|
+
scenarios.add_argument(
|
|
1226
|
+
"--guidance",
|
|
1227
|
+
action="append",
|
|
1228
|
+
default=None,
|
|
1229
|
+
metavar="TEXT",
|
|
1230
|
+
help=(
|
|
1231
|
+
"a natural-language requirement the saved suite must satisfy; repeatable. "
|
|
1232
|
+
"Non-interactive extend/adjust runs pass these to steer the added scenarios "
|
|
1233
|
+
"while preserving existing validated work"
|
|
1234
|
+
),
|
|
1235
|
+
)
|
|
1236
|
+
scenarios.set_defaults(run=_scenarios, interactive=True)
|
|
1237
|
+
|
|
1238
|
+
live = sub.add_parser(
|
|
1239
|
+
"live", help="run the scenarios against the real agent, as a conversation"
|
|
1240
|
+
)
|
|
1241
|
+
live.add_argument("--name", required=True, help="which agent")
|
|
1242
|
+
live.add_argument("--out", default=None, help="artifact directory")
|
|
1243
|
+
live.add_argument(
|
|
1244
|
+
"--once",
|
|
1245
|
+
dest="interactive",
|
|
1246
|
+
action="store_false",
|
|
1247
|
+
help="run unattended instead of staying open",
|
|
1248
|
+
)
|
|
1249
|
+
live.set_defaults(run=_live, interactive=True)
|
|
1250
|
+
|
|
1251
|
+
runs = sub.add_parser("run", help="run the scenarios and grade what happened")
|
|
1252
|
+
runs.add_argument("--name", required=True, help="which agent")
|
|
1253
|
+
runs.add_argument("--out", default=None, help="artifact directory")
|
|
1254
|
+
runs.add_argument(
|
|
1255
|
+
"--target",
|
|
1256
|
+
default="local",
|
|
1257
|
+
choices=target_kinds(),
|
|
1258
|
+
help="where the agent under test runs",
|
|
1259
|
+
)
|
|
1260
|
+
runs.add_argument(
|
|
1261
|
+
"--only", nargs="*", default=None, help="run only these scenarios, by name"
|
|
1262
|
+
)
|
|
1263
|
+
runs.add_argument("--model", default=None, help="model for the run")
|
|
1264
|
+
runs.add_argument(
|
|
1265
|
+
"--quiet",
|
|
1266
|
+
action="store_true",
|
|
1267
|
+
help="only the verdicts, without the conversations as they happen",
|
|
1268
|
+
)
|
|
1269
|
+
runs.set_defaults(run=_run)
|
|
1270
|
+
|
|
1271
|
+
simulation = sub.add_parser(
|
|
1272
|
+
"simulate",
|
|
1273
|
+
help="run through the agent modality and shipped runtime inferred from its contract",
|
|
1274
|
+
)
|
|
1275
|
+
simulation.add_argument("--name", required=True, help="which agent")
|
|
1276
|
+
simulation.add_argument("--out", default=None, help="artifact directory")
|
|
1277
|
+
simulation.add_argument(
|
|
1278
|
+
"--only", nargs="*", default=None, help="run only these scenarios, by name"
|
|
1279
|
+
)
|
|
1280
|
+
simulation.add_argument("--model", default=None, help=argparse.SUPPRESS)
|
|
1281
|
+
simulation.set_defaults(run=_simulate)
|
|
1282
|
+
|
|
1283
|
+
auto = sub.add_parser(
|
|
1284
|
+
"auto",
|
|
1285
|
+
help="from one agent source connection, build everything and run unattended",
|
|
1286
|
+
)
|
|
1287
|
+
auto.add_argument("--path", required=True, help="agent repository")
|
|
1288
|
+
auto.add_argument(
|
|
1289
|
+
"--name",
|
|
1290
|
+
default=None,
|
|
1291
|
+
help="agent name (defaults to the repository folder name)",
|
|
1292
|
+
)
|
|
1293
|
+
auto.add_argument(
|
|
1294
|
+
"--kind", default="repo", choices=supported(), help="how the agent is supplied"
|
|
1295
|
+
)
|
|
1296
|
+
auto.add_argument(
|
|
1297
|
+
"--out",
|
|
1298
|
+
default=None,
|
|
1299
|
+
help="fresh artifact directory (defaults to a unique session directory)",
|
|
1300
|
+
)
|
|
1301
|
+
auto.add_argument("--count", type=int, default=10, help="number of scenarios")
|
|
1302
|
+
auto.add_argument("--model", default=None, help=argparse.SUPPRESS)
|
|
1303
|
+
auto.add_argument("--run-model", default=None, help=argparse.SUPPRESS)
|
|
1304
|
+
auto.set_defaults(run=_auto)
|
|
1305
|
+
|
|
1306
|
+
author = sub.add_parser(
|
|
1307
|
+
"author",
|
|
1308
|
+
help=(
|
|
1309
|
+
"understand an agent, create its logical environment, and write scenarios "
|
|
1310
|
+
"without executing them"
|
|
1311
|
+
),
|
|
1312
|
+
)
|
|
1313
|
+
author.add_argument("--path", required=True, help="agent repository")
|
|
1314
|
+
author.add_argument(
|
|
1315
|
+
"--name",
|
|
1316
|
+
default=None,
|
|
1317
|
+
help="agent name (defaults to the repository folder name)",
|
|
1318
|
+
)
|
|
1319
|
+
author.add_argument(
|
|
1320
|
+
"--kind", default="repo", choices=supported(), help="how the agent is supplied"
|
|
1321
|
+
)
|
|
1322
|
+
author.add_argument(
|
|
1323
|
+
"--out", required=True, help="fresh authoring artifact directory"
|
|
1324
|
+
)
|
|
1325
|
+
author.add_argument("--count", type=int, default=10, help="number of scenarios")
|
|
1326
|
+
author.add_argument("--model", default=chosen_model(), help=argparse.SUPPRESS)
|
|
1327
|
+
author.set_defaults(run=_auto, authoring_only=True, run_model=None)
|
|
1328
|
+
|
|
1329
|
+
chat = sub.add_parser(
|
|
1330
|
+
"chat",
|
|
1331
|
+
help="one conversation: understand, build the world, write the scenarios",
|
|
1332
|
+
)
|
|
1333
|
+
# Nothing is required. Which agent, where it lives and how many scenarios are all things
|
|
1334
|
+
# you say; naming one here is a shortcut back into work already in progress.
|
|
1335
|
+
chat.add_argument("--name", default=None, help=argparse.SUPPRESS)
|
|
1336
|
+
chat.add_argument("--path", default=None, help=argparse.SUPPRESS)
|
|
1337
|
+
chat.add_argument(
|
|
1338
|
+
"--kind", default="repo", choices=supported(), help=argparse.SUPPRESS
|
|
1339
|
+
)
|
|
1340
|
+
chat.add_argument("--out", default=None, help=argparse.SUPPRESS)
|
|
1341
|
+
chat.set_defaults(run=_chat)
|
|
1342
|
+
return parser
|
|
1343
|
+
|
|
1344
|
+
|
|
1345
|
+
def main(argv: list[str] | None = None) -> int:
|
|
1346
|
+
parser = build_parser()
|
|
1347
|
+
args = parser.parse_args(argv)
|
|
1348
|
+
if getattr(args, "run", None) is None:
|
|
1349
|
+
args = parser.parse_args([*(argv or []), "chat"])
|
|
1350
|
+
return asyncio.run(args.run(args))
|
|
1351
|
+
|
|
1352
|
+
|
|
1353
|
+
if __name__ == "__main__":
|
|
1354
|
+
raise SystemExit(main())
|