agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/harness/build.py
ADDED
|
@@ -0,0 +1,385 @@
|
|
|
1
|
+
"""Stage two: build the world the agent's tools run against.
|
|
2
|
+
|
|
3
|
+
Reads the contract stage one produced and builds a database behind the agent's action space,
|
|
4
|
+
then freezes it. The frozen snapshot is the base state every scenario restores from; a scenario
|
|
5
|
+
adds only the rows it additionally needs.
|
|
6
|
+
|
|
7
|
+
The stage stays open, because a world is usually right on the second look. Correcting a handler
|
|
8
|
+
is the next thing said, and the tool is re-run on the spot.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import os
|
|
15
|
+
from collections.abc import Callable
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from .backends import SessionSpec
|
|
20
|
+
from .config import artifact_dir, chosen_model, load_skill, provisioning
|
|
21
|
+
from .contract import AgentContract
|
|
22
|
+
from .session import Stage
|
|
23
|
+
from .world.snapshot import saved as world_saved
|
|
24
|
+
from .world.tools import WORLD_SERVER, world_tools
|
|
25
|
+
|
|
26
|
+
SKILL = "build-environment"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def blockers(
|
|
30
|
+
contract: AgentContract,
|
|
31
|
+
source_root: str = "",
|
|
32
|
+
*,
|
|
33
|
+
external_runtime: bool = False,
|
|
34
|
+
) -> list[str]:
|
|
35
|
+
"""Reasons the real agent implementation cannot be put in a test environment.
|
|
36
|
+
|
|
37
|
+
These are terminal for environment creation. Synthesising a handler would make the suite
|
|
38
|
+
test the harness's interpretation of the agent, so the only honest response is to name the
|
|
39
|
+
missing seam and let the agent owner expose it.
|
|
40
|
+
"""
|
|
41
|
+
# In connect-only provider mode the provider owns the already-deployed tool runtime. The
|
|
42
|
+
# harness must exercise those exact endpoints through the live agent; asking for a local
|
|
43
|
+
# import/service entrypoint would silently turn an optional source upload into a requirement.
|
|
44
|
+
# This exemption is intentionally explicit and is never inferred for repository-backed jobs.
|
|
45
|
+
if external_runtime:
|
|
46
|
+
return []
|
|
47
|
+
|
|
48
|
+
problems: list[str] = []
|
|
49
|
+
if contract.tools and not source_root:
|
|
50
|
+
problems.append(
|
|
51
|
+
"the agent source path was not preserved; reopen the session with the repository "
|
|
52
|
+
"path (or pass --path) so its shipped implementation can be run"
|
|
53
|
+
)
|
|
54
|
+
# A voice worker is itself the runnable seam. Frameworks commonly create function tools as
|
|
55
|
+
# closures that capture session state; forcing those closures to also be importable outside
|
|
56
|
+
# the worker rejects valid agents or encourages the harness to rewrite their behavior. The
|
|
57
|
+
# provisioning stage still has to prove that the submitted worker can actually be packaged.
|
|
58
|
+
runtime_owned = bool(
|
|
59
|
+
contract.modality == "voice"
|
|
60
|
+
and contract.runtime
|
|
61
|
+
and (
|
|
62
|
+
contract.runtime.command
|
|
63
|
+
or contract.runtime.dockerfile
|
|
64
|
+
or contract.runtime.install
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
entries = {entry.tool: entry for entry in contract.tool_entrypoints}
|
|
68
|
+
for tool in contract.tools:
|
|
69
|
+
entry = entries.get(tool.name)
|
|
70
|
+
if entry is None or entry.mode in ("", "generate", "unreachable"):
|
|
71
|
+
if runtime_owned:
|
|
72
|
+
continue
|
|
73
|
+
problems.append(
|
|
74
|
+
f"{tool.name}: no runnable shipped entrypoint was identified; expose the real "
|
|
75
|
+
"implementation as an importable callable or an HTTP service, then point the "
|
|
76
|
+
"harness at that seam"
|
|
77
|
+
)
|
|
78
|
+
continue
|
|
79
|
+
if runtime_owned:
|
|
80
|
+
continue
|
|
81
|
+
if entry.mode in ("import", "construct") and not (
|
|
82
|
+
entry.module and entry.callable
|
|
83
|
+
):
|
|
84
|
+
problems.append(
|
|
85
|
+
f"{tool.name}: {entry.mode} entrypoint needs both module and callable"
|
|
86
|
+
)
|
|
87
|
+
if entry.mode == "service" and not contract.dependencies:
|
|
88
|
+
problems.append(
|
|
89
|
+
f"{tool.name}: service entrypoint has no service dependency describing what "
|
|
90
|
+
"must be started and which configuration points the agent to it"
|
|
91
|
+
)
|
|
92
|
+
return problems
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def require_buildable(
|
|
96
|
+
contract: AgentContract,
|
|
97
|
+
source_root: str = "",
|
|
98
|
+
*,
|
|
99
|
+
external_runtime: bool = False,
|
|
100
|
+
) -> None:
|
|
101
|
+
problems = blockers(
|
|
102
|
+
contract,
|
|
103
|
+
source_root,
|
|
104
|
+
external_runtime=external_runtime,
|
|
105
|
+
)
|
|
106
|
+
if problems:
|
|
107
|
+
raise RuntimeError(
|
|
108
|
+
"Cannot create a truthful test environment without reimplementing agent behavior:\n"
|
|
109
|
+
" - " + "\n - ".join(problems)
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def turns_for(contract: AgentContract) -> int:
|
|
114
|
+
"""A turn budget that grows with the agent being built for.
|
|
115
|
+
|
|
116
|
+
A fixed ceiling silently truncates the work. The budget follows the number of real tool
|
|
117
|
+
bindings that must be exercised rather than a number that happened to fit the first agent.
|
|
118
|
+
"""
|
|
119
|
+
return max(80, len(contract.tools or []) * 8 + 40)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
DEFAULT_HOSTED_ENVIRONMENT_MAX_TURNS = 200
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def environment_turns_for(
|
|
126
|
+
contract: AgentContract, *, requested: int = 0, deferred_runtime: bool = False
|
|
127
|
+
) -> int:
|
|
128
|
+
"""Bound unattended hosted correction without changing local authoring.
|
|
129
|
+
|
|
130
|
+
Local interactive/Compose authoring retains the size-aware budget. Hosted authoring has no
|
|
131
|
+
operator present and must not spend unbounded turns trying variations when the runtime or
|
|
132
|
+
adapter cannot satisfy a validation gate. The ceiling matches what the size-aware budget
|
|
133
|
+
allows a twenty-tool agent, so a normal agent is bounded rather than starved. An explicit
|
|
134
|
+
lower requested budget still wins; an explicit larger one remains capped in the Dockerless
|
|
135
|
+
hosted lane.
|
|
136
|
+
"""
|
|
137
|
+
budget = requested or turns_for(contract)
|
|
138
|
+
if not deferred_runtime:
|
|
139
|
+
return budget
|
|
140
|
+
raw_limit = os.getenv(
|
|
141
|
+
"ALK_HOSTED_ENVIRONMENT_MAX_TURNS",
|
|
142
|
+
str(DEFAULT_HOSTED_ENVIRONMENT_MAX_TURNS),
|
|
143
|
+
)
|
|
144
|
+
try:
|
|
145
|
+
limit = int(raw_limit)
|
|
146
|
+
except ValueError:
|
|
147
|
+
limit = DEFAULT_HOSTED_ENVIRONMENT_MAX_TURNS
|
|
148
|
+
if limit < 1:
|
|
149
|
+
limit = DEFAULT_HOSTED_ENVIRONMENT_MAX_TURNS
|
|
150
|
+
return min(budget, limit)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def open_stage(
|
|
154
|
+
contract: AgentContract,
|
|
155
|
+
*,
|
|
156
|
+
out: Path | None = None,
|
|
157
|
+
ask: Callable[..., Any] | None = None,
|
|
158
|
+
source_root: str = "",
|
|
159
|
+
max_turns: int = 0,
|
|
160
|
+
deferred_runtime: bool = False,
|
|
161
|
+
external_runtime: bool = False,
|
|
162
|
+
) -> tuple[Stage, Path]:
|
|
163
|
+
"""A live build-the-world stage, and where it will write."""
|
|
164
|
+
destination = out or artifact_dir(contract.agent)
|
|
165
|
+
from .provision import ProvisionedEnvironment
|
|
166
|
+
|
|
167
|
+
provisioned = ProvisionedEnvironment.load(destination)
|
|
168
|
+
environment_note = ""
|
|
169
|
+
declared_store = str(
|
|
170
|
+
getattr(getattr(contract, "data_store", None), "kind", "") or ""
|
|
171
|
+
).strip()
|
|
172
|
+
# The authoring store does not enforce the agent's types, so a structured column written as
|
|
173
|
+
# text survives here and is rejected on the first real call.
|
|
174
|
+
declared_store_note = (
|
|
175
|
+
"\n\nThis agent stores its records in "
|
|
176
|
+
+ declared_store
|
|
177
|
+
+ ", and the baseline seeded here is replayed into that engine before the first call. "
|
|
178
|
+
"Seed structured columns as real lists and objects, never as JSON text."
|
|
179
|
+
if declared_store.lower() not in ("", "sqlite", "in_process", "memory")
|
|
180
|
+
else ""
|
|
181
|
+
)
|
|
182
|
+
if provisioned is not None and provisioned.running:
|
|
183
|
+
service_tools = [
|
|
184
|
+
entry.tool for entry in contract.tool_entrypoints if entry.mode == "service"
|
|
185
|
+
]
|
|
186
|
+
runtime_tools = (
|
|
187
|
+
contract.tool_names()
|
|
188
|
+
if provisioned.runtime_services and contract.modality == "voice"
|
|
189
|
+
else [
|
|
190
|
+
entry.tool
|
|
191
|
+
for entry in contract.tool_entrypoints
|
|
192
|
+
if entry.mode in ("import", "construct")
|
|
193
|
+
]
|
|
194
|
+
)
|
|
195
|
+
environment_note = (
|
|
196
|
+
"\n\n## Already provisioned from the agent's repository\n\n"
|
|
197
|
+
f"Compose project: {provisioned.project}\n"
|
|
198
|
+
f"Services: {', '.join(provisioned.services)}\n"
|
|
199
|
+
"Point the agent at these endpoints by changing only these settings:\n"
|
|
200
|
+
+ (
|
|
201
|
+
"\n".join(
|
|
202
|
+
f"- {name}={value}"
|
|
203
|
+
for name, value in sorted(provisioned.overrides.items())
|
|
204
|
+
)
|
|
205
|
+
or (
|
|
206
|
+
"- This is a harness-managed dependency environment; the submitted runtime "
|
|
207
|
+
"already receives its internal datastore connection. Bind importable tools "
|
|
208
|
+
"to the attached real store."
|
|
209
|
+
if provisioned.managed
|
|
210
|
+
else "- No URL override could be inferred. Stop and report the missing config seam."
|
|
211
|
+
)
|
|
212
|
+
)
|
|
213
|
+
+ "\n\nThe source-backed world has already bound these service tools through the "
|
|
214
|
+
"submitted HTTP service: "
|
|
215
|
+
+ (", ".join(service_tools) or "none")
|
|
216
|
+
+ "\nThese tools execute inside the submitted worker and are intentionally not "
|
|
217
|
+
"environment endpoints: "
|
|
218
|
+
+ (", ".join(runtime_tools) or "none")
|
|
219
|
+
+ "\nDo not adopt either group, inspect their source again, or recreate any service "
|
|
220
|
+
"or behavior. Do not use run_env_command for source discovery. The contract already "
|
|
221
|
+
"contains that evidence. Inspect the live data once. Preserve useful repository seed "
|
|
222
|
+
"rows. If the submitted schema is empty or lacks the records needed to exercise the "
|
|
223
|
+
"contract's branches, add a small varied realistic baseline through seed only; never "
|
|
224
|
+
"invent or replace schema, migrations, services, or tool behavior. Avoid placeholder "
|
|
225
|
+
"names/addresses and predictable secrets such as 123456. Scenario-specific people, "
|
|
226
|
+
"credentials and edge states belong in scenario setup rather than the shared base. "
|
|
227
|
+
"Your remaining work is the simulator prompt, observable sub-goals/world checks, "
|
|
228
|
+
+ (
|
|
229
|
+
"check_world, and save_world. Do not declare a build-time sequence for "
|
|
230
|
+
"runtime-internal tools: their stateful ordering is proven by real calls."
|
|
231
|
+
if provisioned.runtime_services and contract.modality == "voice"
|
|
232
|
+
else "one truthful service-backed sequence, check_world, and save_world."
|
|
233
|
+
)
|
|
234
|
+
)
|
|
235
|
+
elif external_runtime:
|
|
236
|
+
environment_note = (
|
|
237
|
+
"\n\n## Existing external provider runtime\n\n"
|
|
238
|
+
"The connected provider agent and its deployed HTTP tools are the runtime under "
|
|
239
|
+
"test. There is no repository or harness-controlled datastore. Do not create a "
|
|
240
|
+
"schema, seed records, adopt or bind tools, declare tool sequences, or invent a "
|
|
241
|
+
"local representation of external state. Build an empty conversation-only world, "
|
|
242
|
+
"write a concrete simulator prompt, and define reusable judged sub-goals from "
|
|
243
|
+
"observable conversation and provider tool events. The real calls, not a local "
|
|
244
|
+
"stand-in, exercise the provider tools."
|
|
245
|
+
)
|
|
246
|
+
elif deferred_runtime:
|
|
247
|
+
environment_note = (
|
|
248
|
+
"\n\n## Runtime deferred to hosted execution\n\n"
|
|
249
|
+
"The submitted repository processes and datastore will be built, started and "
|
|
250
|
+
"validated inside the Daytona execution sandbox from the sealed process bundle. "
|
|
251
|
+
"Do not start containers here. Build the deterministic baseline, simulator prompt, "
|
|
252
|
+
"sub-goals and world checks. Runtime tools are owned by the submitted process for "
|
|
253
|
+
"every modality: do not replace, bind or smoke-call them in this credentialed "
|
|
254
|
+
"control process. Their dependencies are installed in the writable target build "
|
|
255
|
+
"tree and their real behavior is validated when generated scenarios run in Daytona."
|
|
256
|
+
+ declared_store_note
|
|
257
|
+
+ "\n\nYou cannot call a tool here, so you have not seen a single real response "
|
|
258
|
+
"shape. Write every world check and sub-goal against world state you can read now, "
|
|
259
|
+
"never against the fields or wording you expect a tool response to carry. A check "
|
|
260
|
+
"that asserts on an imagined response passes or fails for the wrong reason and "
|
|
261
|
+
"reports the agent did nothing when it did."
|
|
262
|
+
)
|
|
263
|
+
server, _world = world_tools(
|
|
264
|
+
contract,
|
|
265
|
+
destination,
|
|
266
|
+
source_root=source_root,
|
|
267
|
+
deferred_runtime=deferred_runtime,
|
|
268
|
+
external_runtime=external_runtime,
|
|
269
|
+
)
|
|
270
|
+
spec = SessionSpec(
|
|
271
|
+
system_prompt=(
|
|
272
|
+
f"{load_skill(SKILL)}\n\n## This agent\n\n{contract.brief(with_data=True)}"
|
|
273
|
+
+ environment_note
|
|
274
|
+
),
|
|
275
|
+
# No file tools and no shell. Everything this stage can do goes through a tool that
|
|
276
|
+
# executes it and reports back, which is what makes the guardrails meaningful.
|
|
277
|
+
servers={WORLD_SERVER: server},
|
|
278
|
+
builtins=("AskUserQuestion",),
|
|
279
|
+
cwd=str(destination.parent if destination.parent.exists() else Path.cwd()),
|
|
280
|
+
max_turns=environment_turns_for(
|
|
281
|
+
contract,
|
|
282
|
+
requested=max_turns,
|
|
283
|
+
deferred_runtime=deferred_runtime,
|
|
284
|
+
),
|
|
285
|
+
model=chosen_model(),
|
|
286
|
+
ask=ask,
|
|
287
|
+
thinking=True,
|
|
288
|
+
)
|
|
289
|
+
return Stage(spec, name=SKILL), destination
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def opening(
|
|
293
|
+
contract: AgentContract,
|
|
294
|
+
*,
|
|
295
|
+
provisioned: bool = False,
|
|
296
|
+
deferred_runtime: bool = False,
|
|
297
|
+
external_runtime: bool = False,
|
|
298
|
+
) -> str:
|
|
299
|
+
if provisioning():
|
|
300
|
+
return (
|
|
301
|
+
f"Provision the environment for {contract.agent!r}.\n\n"
|
|
302
|
+
"Stand up the engine it already uses, run its OWN migrations into it, and seed "
|
|
303
|
+
"from the contract's real data. Do not write a schema and do not touch the "
|
|
304
|
+
"agent — you are building what it connects to, not a copy of it."
|
|
305
|
+
)
|
|
306
|
+
if provisioned:
|
|
307
|
+
sequence_instruction = (
|
|
308
|
+
"Do not declare or smoke-call runtime-internal tools outside their voice session; "
|
|
309
|
+
"the real scenarios prove their stateful ordering. "
|
|
310
|
+
if contract.modality == "voice" and contract.runtime
|
|
311
|
+
else "Declare one sequence using already-bound service endpoints with real required "
|
|
312
|
+
"arguments. "
|
|
313
|
+
)
|
|
314
|
+
return (
|
|
315
|
+
f"Finish the already-provisioned source-backed world for {contract.agent!r}.\n\n"
|
|
316
|
+
"The submitted Compose services, seed data, HTTP service-tool bindings, and "
|
|
317
|
+
"worker-internal tools are already authoritative and must not be rebuilt or adopted. "
|
|
318
|
+
"Inspect existing state once; keep useful submitted seed rows, and only seed missing "
|
|
319
|
+
"baseline records into existing collections when the world would otherwise be too "
|
|
320
|
+
"empty to exercise the contract. Use varied realistic values, never demo placeholders "
|
|
321
|
+
"or predictable credentials. Write the simulator prompt, add "
|
|
322
|
+
"observable sub-goals and world checks. "
|
|
323
|
+
+ sequence_instruction
|
|
324
|
+
+ "Then check_world and save_world. "
|
|
325
|
+
"Do not read source, run shell commands, or investigate runtime-internal tools."
|
|
326
|
+
)
|
|
327
|
+
if external_runtime:
|
|
328
|
+
return (
|
|
329
|
+
f"Build the black-box test world for {contract.agent!r}.\n\n"
|
|
330
|
+
"Keep it empty: the existing provider agent owns its tools and external state. "
|
|
331
|
+
"Do not seed a shadow database or replay provider tools here. Add shared judged "
|
|
332
|
+
"conversation/tool-observation sub-goals, write the simulator prompt, check the "
|
|
333
|
+
"empty world, and save it."
|
|
334
|
+
)
|
|
335
|
+
if deferred_runtime:
|
|
336
|
+
return (
|
|
337
|
+
f"Build the logical baseline for {contract.agent!r}.\n\n"
|
|
338
|
+
"The submitted runtime and all of its tools are built later by the isolated process "
|
|
339
|
+
"runtime. Do not adopt or smoke-call those tools here. Seed the source-backed data "
|
|
340
|
+
"described by the contract, write the simulator prompt, add observable sub-goals "
|
|
341
|
+
"and world checks, then check_world and save_world. Runtime tool coverage and "
|
|
342
|
+
"stateful ordering are proven by the real generated scenarios."
|
|
343
|
+
)
|
|
344
|
+
return (
|
|
345
|
+
f"Build the world for {contract.agent!r}.\n\n"
|
|
346
|
+
"Use only the runtime, services, migrations, data loaders and tool implementations the "
|
|
347
|
+
"agent ships. Bind one handler per tool to each real implementation and verify its "
|
|
348
|
+
"refusals with run_tool. If any real "
|
|
349
|
+
"implementation cannot be reached, state the exact missing seam and stop; never write a "
|
|
350
|
+
"replacement. Declare at least one stateful sequence, then check_world and save_world."
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
async def build(
|
|
355
|
+
contract: AgentContract,
|
|
356
|
+
*,
|
|
357
|
+
out: Path | None = None,
|
|
358
|
+
source_root: str = "",
|
|
359
|
+
follow_ups: list[str] | None = None,
|
|
360
|
+
on_event: Callable[..., Any] | None = None,
|
|
361
|
+
ask: Callable[..., Any] | None = None,
|
|
362
|
+
max_turns: int = 0,
|
|
363
|
+
) -> Path | None:
|
|
364
|
+
"""Run the stage start to finish. Returns where the world was written, or None."""
|
|
365
|
+
require_buildable(contract, source_root)
|
|
366
|
+
destination = out or artifact_dir(contract.agent)
|
|
367
|
+
from .provision import provision_if_present
|
|
368
|
+
|
|
369
|
+
provisioned = await asyncio.to_thread(
|
|
370
|
+
provision_if_present, source_root, destination, contract
|
|
371
|
+
)
|
|
372
|
+
stage, destination = open_stage(
|
|
373
|
+
contract,
|
|
374
|
+
out=destination,
|
|
375
|
+
ask=ask,
|
|
376
|
+
source_root=source_root,
|
|
377
|
+
max_turns=max_turns,
|
|
378
|
+
)
|
|
379
|
+
async with stage:
|
|
380
|
+
await stage.say(
|
|
381
|
+
opening(contract, provisioned=provisioned is not None), on_event=on_event
|
|
382
|
+
)
|
|
383
|
+
for follow_up in follow_ups or []:
|
|
384
|
+
await stage.say(follow_up, on_event=on_event)
|
|
385
|
+
return destination if world_saved(destination) else None
|