agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,718 @@
|
|
|
1
|
+
"""The agent contract: what the agent verifiably is, read from its own source.
|
|
2
|
+
|
|
3
|
+
Everything downstream is confined to this. A world may only implement tools listed here, a
|
|
4
|
+
scenario may only reference values grounded in here, and a checkpoint may only assert against
|
|
5
|
+
what is here. It is the anti-hallucination device for every later stage.
|
|
6
|
+
|
|
7
|
+
The harness produces it by reading the agent's code and calling ``submit_contract``. Validation
|
|
8
|
+
runs inside that tool, so problems are returned into the conversation and the model tries again
|
|
9
|
+
rather than a bad contract reaching disk.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import shlex
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel, Field, field_validator, model_validator
|
|
19
|
+
|
|
20
|
+
# How a person reaches an agent. This decides how it is later run — voice goes out as a live
|
|
21
|
+
# call, everything else runs locally — so it is defined once and referenced, never retyped.
|
|
22
|
+
MODALITIES = ("voice", "chat", "browser")
|
|
23
|
+
# Voice only, and only two: either the agent placed the call or it answered one.
|
|
24
|
+
CALL_DIRECTIONS = ("inbound", "outbound")
|
|
25
|
+
|
|
26
|
+
_STRING_FIELDS = (
|
|
27
|
+
"agent",
|
|
28
|
+
"one_liner",
|
|
29
|
+
"modality",
|
|
30
|
+
"call_direction",
|
|
31
|
+
"system_prompt_excerpt",
|
|
32
|
+
"notes",
|
|
33
|
+
)
|
|
34
|
+
_LIST_FIELDS = (
|
|
35
|
+
"hard_constraints",
|
|
36
|
+
"real_use_cases",
|
|
37
|
+
"amendments",
|
|
38
|
+
"chosen_evals",
|
|
39
|
+
)
|
|
40
|
+
_DICT_FIELDS = ("data_schema", "base_environment")
|
|
41
|
+
|
|
42
|
+
# What each field gets called when it is not called what we call it. Every one of these was
|
|
43
|
+
# written by a model that had read the schema and still reached for the more obvious word.
|
|
44
|
+
_ALIASES = {
|
|
45
|
+
"real_use_cases": ("use_cases", "usecases", "scenarios", "capabilities"),
|
|
46
|
+
"hard_constraints": ("constraints", "rules", "policies", "policy", "guardrails"),
|
|
47
|
+
"system_prompt_excerpt": ("system_prompt", "prompt", "instructions"),
|
|
48
|
+
"base_environment": ("data", "seed_data", "starting_data", "records"),
|
|
49
|
+
"data_schema": ("schema", "record_schema", "data_shape"),
|
|
50
|
+
"agent": ("name", "agent_name"),
|
|
51
|
+
"one_liner": ("summary", "description"),
|
|
52
|
+
"notes": ("observations", "remarks"),
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class ToolSpec(BaseModel):
|
|
57
|
+
"""One tool the agent really has.
|
|
58
|
+
|
|
59
|
+
``args`` is the load-bearing field: the world's handlers, the probes and every scenario are
|
|
60
|
+
built from these exact names. It is also the one most often written under another name —
|
|
61
|
+
``parameters``, ``arguments``, ``params`` — or left out while ``arg_types`` names every
|
|
62
|
+
argument anyway. All of those are the same information, so they are accepted and normalised
|
|
63
|
+
rather than rejected, because a contract bounced for a synonym costs a full turn and teaches
|
|
64
|
+
nothing about the agent.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
@model_validator(mode="before")
|
|
68
|
+
@classmethod
|
|
69
|
+
def _normalize_args(cls, payload: Any) -> Any:
|
|
70
|
+
if not isinstance(payload, dict):
|
|
71
|
+
return payload
|
|
72
|
+
if not payload.get("args"):
|
|
73
|
+
for alias in ("parameters", "arguments", "params", "arg_names"):
|
|
74
|
+
value = payload.get(alias)
|
|
75
|
+
if isinstance(value, list) and value:
|
|
76
|
+
payload["args"] = value
|
|
77
|
+
break
|
|
78
|
+
# Some writers give {name: type} where a list was asked for. The keys are the
|
|
79
|
+
# argument names, which is exactly what was wanted.
|
|
80
|
+
if isinstance(value, dict) and value:
|
|
81
|
+
payload["args"] = list(value)
|
|
82
|
+
payload.setdefault(
|
|
83
|
+
"arg_types", {k: str(v) for k, v in value.items()}
|
|
84
|
+
)
|
|
85
|
+
break
|
|
86
|
+
if not payload.get("args"):
|
|
87
|
+
# Nothing named the arguments directly, but a per-argument map still names them.
|
|
88
|
+
for source in ("arg_types", "arg_values"):
|
|
89
|
+
mapping = payload.get(source)
|
|
90
|
+
if isinstance(mapping, dict) and mapping:
|
|
91
|
+
payload["args"] = list(mapping)
|
|
92
|
+
break
|
|
93
|
+
if isinstance(payload.get("args"), str):
|
|
94
|
+
payload["args"] = [payload["args"]]
|
|
95
|
+
if isinstance(payload.get("args"), list):
|
|
96
|
+
payload["args"] = [str(one) for one in payload["args"]]
|
|
97
|
+
return payload
|
|
98
|
+
|
|
99
|
+
name: str
|
|
100
|
+
args: list[str] = Field(default_factory=list)
|
|
101
|
+
arg_types: dict[str, str] = Field(default_factory=dict)
|
|
102
|
+
arg_values: dict[str, Any] = Field(default_factory=dict)
|
|
103
|
+
description: str = ""
|
|
104
|
+
# Tools that must have run before this one stops refusing, and only for state the agent builds
|
|
105
|
+
# during the conversation. Empty means callable first thing. Marking a tool gated when it is not
|
|
106
|
+
# costs every future test of it a preamble it never needed.
|
|
107
|
+
requires: list[str] = Field(default_factory=list)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class ToolEntry(BaseModel):
|
|
111
|
+
"""How to reach the agent's own implementation of one tool.
|
|
112
|
+
|
|
113
|
+
Recorded rather than assumed, because there is no shape every agent shares. A benchmark
|
|
114
|
+
writes static methods on a class; a framework agent writes closures inside ``__init__`` that
|
|
115
|
+
cannot be imported at all. What the environment does about a tool is decided from ``mode``,
|
|
116
|
+
so a tool nobody can reach is visible here rather than quietly reimplemented.
|
|
117
|
+
"""
|
|
118
|
+
|
|
119
|
+
tool: str
|
|
120
|
+
# import: a module-level callable. construct: a method needing an instance built first.
|
|
121
|
+
# service: reachable over HTTP. unreachable: no runnable seam was found and building stops.
|
|
122
|
+
# The harness never generates agent behavior.
|
|
123
|
+
mode: str = "unreachable"
|
|
124
|
+
module: str = ""
|
|
125
|
+
callable: str = ""
|
|
126
|
+
# An expression that builds the object a `construct` tool hangs off.
|
|
127
|
+
factory: str = ""
|
|
128
|
+
# What the agent's own state is passed as, where a tool takes it as an argument.
|
|
129
|
+
first_arg: str = ""
|
|
130
|
+
# For a service-backed tool, the submitted service and HTTP path its implementation calls.
|
|
131
|
+
# The path is recorded separately from the semantic tool name because production APIs often
|
|
132
|
+
# use a different route name (for example check_status -> /get_status).
|
|
133
|
+
service: str = ""
|
|
134
|
+
endpoint: str = ""
|
|
135
|
+
method: str = "POST"
|
|
136
|
+
notes: str = ""
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class DataStore(BaseModel):
|
|
140
|
+
"""What the agent's tools read and write, and how to be there instead of it.
|
|
141
|
+
|
|
142
|
+
Nothing recorded here is a change to the agent. It is what the agent **already expects**,
|
|
143
|
+
written down so the environment can be built to match: the same host, the same port, the same
|
|
144
|
+
database, the same user. Where it reads a value from configuration we set that configuration;
|
|
145
|
+
where it hardcodes one we shape our own store to it, which is why a hardcoded value is worth
|
|
146
|
+
recording rather than treated as a dead end.
|
|
147
|
+
|
|
148
|
+
That inversion is the point. The alternative, editing the agent until it points at us, means
|
|
149
|
+
testing something other than what ships.
|
|
150
|
+
"""
|
|
151
|
+
|
|
152
|
+
# Read off the agent, never chosen for it. Postgres and ClickHouse disagree about dialect,
|
|
153
|
+
# types and what a transaction even means, so an agent tested against the wrong one is graded
|
|
154
|
+
# on queries it never runs. Free text because the next agent will be on an engine nobody has
|
|
155
|
+
# written down yet.
|
|
156
|
+
kind: str = ""
|
|
157
|
+
version: str = ""
|
|
158
|
+
|
|
159
|
+
# The easiest seam, and the one most agents have: one variable or config key holding the whole
|
|
160
|
+
# connection string. Set it at launch and nothing else matters.
|
|
161
|
+
configured_by: str = ""
|
|
162
|
+
config_key: str = ""
|
|
163
|
+
|
|
164
|
+
# What the agent expects to find, whether it reads these from config or has them written into
|
|
165
|
+
# its source. A hardcoded host is not an obstacle: a network alias makes that name resolve to
|
|
166
|
+
# our container, and the agent connects to us believing nothing changed.
|
|
167
|
+
host: str = ""
|
|
168
|
+
port: int | None = None
|
|
169
|
+
database: str = ""
|
|
170
|
+
user: str = ""
|
|
171
|
+
# Deliberately never the password itself. A contract is written to disk and read by people, so
|
|
172
|
+
# a secret in it outlives the run that needed it. What is recorded is where the value comes
|
|
173
|
+
# from; if it is genuinely needed it is read at build time and not persisted.
|
|
174
|
+
password_from: str = ""
|
|
175
|
+
|
|
176
|
+
# An agent that holds its data in memory is reached by calling the function that loads it, not
|
|
177
|
+
# by connecting to anything. Recorded so the environment can call the agent's own loader
|
|
178
|
+
# rather than reading its files and rebuilding the structure itself, which would be a second
|
|
179
|
+
# implementation of the one thing this path exists to stop reimplementing.
|
|
180
|
+
schema_from: str = ""
|
|
181
|
+
loaded_by: str = ""
|
|
182
|
+
loader_module: str = ""
|
|
183
|
+
|
|
184
|
+
def has_seam(self) -> bool:
|
|
185
|
+
"""Whether there is any way to point this agent at our store.
|
|
186
|
+
|
|
187
|
+
An agent with no seam at all is a finding, not a thing to work around: it cannot be tested
|
|
188
|
+
without one, and saying so is more useful than editing it until it can.
|
|
189
|
+
"""
|
|
190
|
+
return bool(
|
|
191
|
+
self.configured_by
|
|
192
|
+
or self.config_key
|
|
193
|
+
or self.host
|
|
194
|
+
or self.port
|
|
195
|
+
or self.database
|
|
196
|
+
or self.loader_module
|
|
197
|
+
or self.loaded_by
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
class Reached(BaseModel):
|
|
202
|
+
"""How an existing agent reaches a dependency, without storing its secret values."""
|
|
203
|
+
|
|
204
|
+
dsn_env: str = ""
|
|
205
|
+
config_key: str = ""
|
|
206
|
+
host: str = ""
|
|
207
|
+
port: int | None = None
|
|
208
|
+
database: str = ""
|
|
209
|
+
user: str = ""
|
|
210
|
+
password_from: str = ""
|
|
211
|
+
loader_module: str = ""
|
|
212
|
+
loader_function: str = ""
|
|
213
|
+
|
|
214
|
+
def has_seam(self) -> bool:
|
|
215
|
+
return bool(
|
|
216
|
+
self.dsn_env
|
|
217
|
+
or self.config_key
|
|
218
|
+
or self.host
|
|
219
|
+
or self.port
|
|
220
|
+
or self.database
|
|
221
|
+
or self.loader_module
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class RuntimeInterface(BaseModel):
|
|
226
|
+
"""The submitted runtime's existing conversational ingress.
|
|
227
|
+
|
|
228
|
+
This is connection metadata, not generated agent behavior. The harness may publish the
|
|
229
|
+
declared container port and translate its request/response envelope, but it never adds an
|
|
230
|
+
endpoint the repository does not already implement.
|
|
231
|
+
"""
|
|
232
|
+
|
|
233
|
+
kind: str = ""
|
|
234
|
+
protocol: str = "fi.alk"
|
|
235
|
+
port: int | None = Field(default=None, ge=1, le=65535)
|
|
236
|
+
path: str = ""
|
|
237
|
+
health_path: str = ""
|
|
238
|
+
include_tools: bool = True
|
|
239
|
+
|
|
240
|
+
@field_validator("kind")
|
|
241
|
+
@classmethod
|
|
242
|
+
def _known_kind(cls, value: str) -> str:
|
|
243
|
+
normalized = str(value or "").strip().lower().replace("-", "_")
|
|
244
|
+
aliases = {"openai": "http", "openai_compatible": "http"}
|
|
245
|
+
return aliases.get(normalized, normalized)
|
|
246
|
+
|
|
247
|
+
@field_validator("protocol")
|
|
248
|
+
@classmethod
|
|
249
|
+
def _known_protocol(cls, value: str) -> str:
|
|
250
|
+
normalized = str(value or "fi.alk").strip().lower().replace("-", "_")
|
|
251
|
+
aliases = {
|
|
252
|
+
"openai": "openai_chat",
|
|
253
|
+
"openai_compatible": "openai_chat",
|
|
254
|
+
"chat_completions": "openai_chat",
|
|
255
|
+
"http": "fi.alk",
|
|
256
|
+
}
|
|
257
|
+
return aliases.get(normalized, normalized)
|
|
258
|
+
|
|
259
|
+
@field_validator("path", "health_path")
|
|
260
|
+
@classmethod
|
|
261
|
+
def _absolute_http_path(cls, value: str) -> str:
|
|
262
|
+
path = str(value or "").strip()
|
|
263
|
+
if path and not path.startswith("/"):
|
|
264
|
+
path = "/" + path
|
|
265
|
+
return path
|
|
266
|
+
|
|
267
|
+
@model_validator(mode="after")
|
|
268
|
+
def _complete(self) -> "RuntimeInterface":
|
|
269
|
+
if self.kind in {"http", "websocket"}:
|
|
270
|
+
if self.port is None:
|
|
271
|
+
raise ValueError(f"runtime_{self.kind}_interface_requires_port")
|
|
272
|
+
if not self.path:
|
|
273
|
+
raise ValueError(f"runtime_{self.kind}_interface_requires_path")
|
|
274
|
+
if self.kind == "http":
|
|
275
|
+
if self.protocol not in {"fi.alk", "openai_chat"}:
|
|
276
|
+
raise ValueError(
|
|
277
|
+
"runtime_http_protocol_unsupported: expected fi.alk or openai_chat"
|
|
278
|
+
)
|
|
279
|
+
if self.kind == "websocket" and self.protocol != "fi.alk":
|
|
280
|
+
raise ValueError("runtime_websocket_protocol_unsupported: expected fi.alk")
|
|
281
|
+
return self
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
class Runtime(BaseModel):
|
|
285
|
+
"""What it takes to run the agent's code."""
|
|
286
|
+
|
|
287
|
+
# Empty means detect from the submitted dependency manifest. Defaulting this to Python makes
|
|
288
|
+
# an otherwise unambiguous Node repository fail the generated-runtime admission path.
|
|
289
|
+
language: str = ""
|
|
290
|
+
version: str = ""
|
|
291
|
+
install: str = ""
|
|
292
|
+
# Optional dependency groups declared by the repository itself (for example ``voice`` in
|
|
293
|
+
# pyproject.toml). Generated packaging validates these names against the manifest.
|
|
294
|
+
extras: list[str] = Field(default_factory=list)
|
|
295
|
+
workdir: str = ""
|
|
296
|
+
# Select one submitted Compose file when a repository contains multiple runnable stacks.
|
|
297
|
+
compose_file: str = ""
|
|
298
|
+
dockerfile: str = ""
|
|
299
|
+
# For repositories without container metadata, this is an argv vector for the submitted
|
|
300
|
+
# process. It is optional when one conventional entrypoint can be proven from source.
|
|
301
|
+
command: list[str] = Field(default_factory=list)
|
|
302
|
+
# Repository-relative generated-build exclusions selected during understanding. This is for
|
|
303
|
+
# large checked-in outputs or documentation, never for dependency manifests or source code.
|
|
304
|
+
context_excludes: list[str] = Field(default_factory=list)
|
|
305
|
+
# Preserve a repository-declared target architecture (for example linux/amd64 on an ARM
|
|
306
|
+
# runner). This is execution metadata, not a change to the submitted application.
|
|
307
|
+
platform: str = ""
|
|
308
|
+
|
|
309
|
+
# How a turn-based simulator reaches the submitted process after it starts. Voice has a
|
|
310
|
+
# standard rendezvous (LiveKit dispatch); chat repositories do not. Recording this seam is
|
|
311
|
+
# what lets the harness start the real runtime instead of reconstructing the agent from its
|
|
312
|
+
# prompt. Empty is valid for voice/browser agents and for contracts that are not backed by a
|
|
313
|
+
# repository.
|
|
314
|
+
interface: RuntimeInterface | None = None
|
|
315
|
+
|
|
316
|
+
@field_validator("command", mode="before")
|
|
317
|
+
@classmethod
|
|
318
|
+
def _normalize_command(cls, value: Any) -> Any:
|
|
319
|
+
if isinstance(value, str):
|
|
320
|
+
return shlex.split(value)
|
|
321
|
+
return value
|
|
322
|
+
|
|
323
|
+
@field_validator("extras", mode="before")
|
|
324
|
+
@classmethod
|
|
325
|
+
def _normalize_extras(cls, value: Any) -> Any:
|
|
326
|
+
if isinstance(value, str):
|
|
327
|
+
return [item.strip() for item in value.split(",") if item.strip()]
|
|
328
|
+
return value
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
class Dependency(BaseModel):
|
|
332
|
+
"""Something the agent reaches for that has to exist before it can work.
|
|
333
|
+
|
|
334
|
+
This is what tells the environment stage there is a service to stand up, rather than leaving
|
|
335
|
+
it to notice halfway through that a tool has nothing to answer it. The world is a sandbox:
|
|
336
|
+
whatever is named here gets built inside it, so the agent's call goes to something real that
|
|
337
|
+
happens to be ours.
|
|
338
|
+
"""
|
|
339
|
+
|
|
340
|
+
name: str
|
|
341
|
+
# datastore, service, file, queue — whatever kind of thing this is. Left open rather than
|
|
342
|
+
# enumerated, because the next agent will need a kind nobody has thought of yet.
|
|
343
|
+
kind: str = ""
|
|
344
|
+
what: str = ""
|
|
345
|
+
# The tools that cannot work without it. An unreferenced dependency is usually a mistake.
|
|
346
|
+
used_by: list[str] = Field(default_factory=list)
|
|
347
|
+
engine: str = ""
|
|
348
|
+
version: str = ""
|
|
349
|
+
reached: Reached = Field(default_factory=Reached)
|
|
350
|
+
|
|
351
|
+
def provisionable(self) -> bool:
|
|
352
|
+
return bool(self.engine) and self.reached.has_seam()
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def _reached(one: Dependency) -> str:
|
|
356
|
+
if not one.engine and not one.reached.has_seam():
|
|
357
|
+
return ""
|
|
358
|
+
said: list[str] = []
|
|
359
|
+
if one.engine:
|
|
360
|
+
said.append(f"stand up {one.engine}{' ' + one.version if one.version else ''}")
|
|
361
|
+
where = one.reached
|
|
362
|
+
if where.loader_module or where.loader_function:
|
|
363
|
+
said.append(
|
|
364
|
+
"call the agent's own "
|
|
365
|
+
f"{where.loader_module or 'MODULE NOT RECORDED'}."
|
|
366
|
+
f"{where.loader_function or 'load_data'} for it; nothing is connected to and no "
|
|
367
|
+
"server is involved"
|
|
368
|
+
)
|
|
369
|
+
elif where.dsn_env:
|
|
370
|
+
said.append(f"point it there with ${where.dsn_env}")
|
|
371
|
+
elif where.config_key:
|
|
372
|
+
said.append(f"point it there with {where.config_key} in its config")
|
|
373
|
+
expected = [
|
|
374
|
+
f"{label} {value}"
|
|
375
|
+
for label, value in (
|
|
376
|
+
("host", where.host),
|
|
377
|
+
("port", where.port),
|
|
378
|
+
("database", where.database),
|
|
379
|
+
("user", where.user),
|
|
380
|
+
)
|
|
381
|
+
if value
|
|
382
|
+
]
|
|
383
|
+
if expected:
|
|
384
|
+
said.append("build it to match " + ", ".join(expected))
|
|
385
|
+
if one.engine and not where.has_seam():
|
|
386
|
+
said.append("NO CONFIGURATION SEAM RECORDED")
|
|
387
|
+
return "; ".join(said)
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
class AgentContract(BaseModel):
|
|
391
|
+
"""What the agent verifiably is. Nothing downstream may contradict this."""
|
|
392
|
+
|
|
393
|
+
@model_validator(mode="before")
|
|
394
|
+
@classmethod
|
|
395
|
+
def _normalize_shapes(cls, payload: Any) -> Any:
|
|
396
|
+
"""Model JSON varies in benign ways: a list where prose was asked, a bare string where a
|
|
397
|
+
list was, a field under the obvious name rather than ours. Normalize instead of
|
|
398
|
+
rejecting, because none of that is a grounding error and rejecting it burns turns on
|
|
399
|
+
something that does not matter."""
|
|
400
|
+
if not isinstance(payload, dict):
|
|
401
|
+
return payload
|
|
402
|
+
# The name we chose is not always the obvious one. `real_use_cases` in particular gets
|
|
403
|
+
# written as `use_cases`, and the answer it then gets — "no-use-cases" — reads as
|
|
404
|
+
# missing rather than misnamed, so the same submission comes back again and again with
|
|
405
|
+
# the shape changed and the name untouched.
|
|
406
|
+
for ours, others in _ALIASES.items():
|
|
407
|
+
if payload.get(ours):
|
|
408
|
+
continue
|
|
409
|
+
for other in others:
|
|
410
|
+
if payload.get(other):
|
|
411
|
+
payload[ours] = payload[other]
|
|
412
|
+
break
|
|
413
|
+
for key in _STRING_FIELDS:
|
|
414
|
+
value = payload.get(key)
|
|
415
|
+
if isinstance(value, list):
|
|
416
|
+
payload[key] = "\n".join(str(item) for item in value)
|
|
417
|
+
elif value is not None and not isinstance(value, str):
|
|
418
|
+
payload[key] = str(value)
|
|
419
|
+
for key in _LIST_FIELDS:
|
|
420
|
+
value = payload.get(key)
|
|
421
|
+
if isinstance(value, str):
|
|
422
|
+
payload[key] = [value]
|
|
423
|
+
elif isinstance(value, list):
|
|
424
|
+
payload[key] = [
|
|
425
|
+
str(item) if not isinstance(item, str) else item for item in value
|
|
426
|
+
]
|
|
427
|
+
for key in _DICT_FIELDS:
|
|
428
|
+
value = payload.get(key)
|
|
429
|
+
if value is not None and not isinstance(value, dict):
|
|
430
|
+
payload[key] = {"value": value}
|
|
431
|
+
return payload
|
|
432
|
+
|
|
433
|
+
# Defaulted rather than mandatory so a submission that forgets it reaches validate_contract,
|
|
434
|
+
# which says what to do about it, instead of dying in the schema layer with a type error.
|
|
435
|
+
agent: str = ""
|
|
436
|
+
one_liner: str = ""
|
|
437
|
+
modality: str = "chat"
|
|
438
|
+
# Voice only: whether this agent places calls or answers them. Chat is always started by the
|
|
439
|
+
# person, so it stays inbound. It changes how the simulated person is briefed, not who speaks
|
|
440
|
+
# first: an outbound agent still greets, it just has to say who it is and why it called.
|
|
441
|
+
call_direction: str = "inbound"
|
|
442
|
+
conversational: bool = True
|
|
443
|
+
system_prompt_excerpt: str = ""
|
|
444
|
+
hard_constraints: list[str] = Field(default_factory=list)
|
|
445
|
+
tools: list[ToolSpec] = Field(default_factory=list)
|
|
446
|
+
data_schema: dict[str, Any] = Field(default_factory=dict)
|
|
447
|
+
base_environment: dict[str, Any] = Field(default_factory=dict)
|
|
448
|
+
# What the environment stage has to build before any tool can be answered.
|
|
449
|
+
dependencies: list[Dependency] = Field(default_factory=list)
|
|
450
|
+
# Transport/model connections are configuration, not business services to reconstruct.
|
|
451
|
+
runtime_dependencies: list[Dependency] = Field(default_factory=list)
|
|
452
|
+
# Whether the agent ships code for its tools: present, absent, or partial. Missing code is a
|
|
453
|
+
# build blocker: the harness never supplies replacement agent behavior.
|
|
454
|
+
implementation: str = ""
|
|
455
|
+
tool_entrypoints: list[ToolEntry] = Field(default_factory=list)
|
|
456
|
+
# How this agent's tools say no in a value they return, rather than by raising. Without it a
|
|
457
|
+
# refusal cannot be told from a success once the agent's own code is answering the call.
|
|
458
|
+
refusal_signature: str = ""
|
|
459
|
+
data_store: DataStore | None = None
|
|
460
|
+
runtime: Runtime | None = None
|
|
461
|
+
real_use_cases: list[str] = Field(default_factory=list)
|
|
462
|
+
# Free-form. The fields above are the fixed core because code consumes them; this is where
|
|
463
|
+
# the reader records whatever else about *this* agent is worth carrying forward — quirks,
|
|
464
|
+
# traps, names that look real but are not — in whatever form fits. It is shown verbatim to
|
|
465
|
+
# every later stage.
|
|
466
|
+
notes: str = ""
|
|
467
|
+
open_questions: list[str] = Field(default_factory=list)
|
|
468
|
+
# Names only; the platform owns the catalogue. Empty means judge by the scenarios' checks alone.
|
|
469
|
+
chosen_evals: list[str] = Field(default_factory=list)
|
|
470
|
+
# Anything in here was not read from the agent's source. The contract is meant to be what
|
|
471
|
+
# the agent verifiably is, so when the harness widens it the difference is recorded rather
|
|
472
|
+
# than blended in, and whoever reads it later can tell the two apart.
|
|
473
|
+
amendments: list[str] = Field(default_factory=list)
|
|
474
|
+
|
|
475
|
+
def tool_names(self) -> set[str]:
|
|
476
|
+
return {tool.name for tool in self.tools}
|
|
477
|
+
|
|
478
|
+
def brief(self, *, full_schema: bool = True, with_data: bool = False) -> str:
|
|
479
|
+
"""The grounding block handed to the model on every downstream call.
|
|
480
|
+
|
|
481
|
+
``with_data`` includes the agent's real starting records rather than only their shape.
|
|
482
|
+
A stage that writes scenarios needs to know a menu exists; a stage that builds the world
|
|
483
|
+
has to reproduce it row for row, and a shape without records is not enough to do that.
|
|
484
|
+
"""
|
|
485
|
+
lines: list[str] = []
|
|
486
|
+
for tool in self.tools:
|
|
487
|
+
signature = ", ".join(
|
|
488
|
+
f"{arg}: {tool.arg_types[arg]}" if arg in tool.arg_types else arg
|
|
489
|
+
for arg in tool.args
|
|
490
|
+
)
|
|
491
|
+
values = (
|
|
492
|
+
f" [values: {json.dumps(tool.arg_values)[:300]}]"
|
|
493
|
+
if tool.arg_values
|
|
494
|
+
else ""
|
|
495
|
+
)
|
|
496
|
+
# Preconditions belong on the tool line or they are not read. A writer that cannot see
|
|
497
|
+
# what a tool refuses until another has run replays the agent's whole flow to reach it.
|
|
498
|
+
needs = f" [after: {', '.join(tool.requires)}]" if tool.requires else ""
|
|
499
|
+
lines.append(
|
|
500
|
+
f" - {tool.name}({signature}){values}{needs} : {tool.description[:140]}"
|
|
501
|
+
)
|
|
502
|
+
parts = [
|
|
503
|
+
f"AGENT: {self.agent} - {self.one_liner}",
|
|
504
|
+
f"MODALITY: {self.modality}",
|
|
505
|
+
"REAL TOOLS (use ONLY these, with these exact arg names and types):\n"
|
|
506
|
+
+ ("\n".join(lines) or " (none)"),
|
|
507
|
+
]
|
|
508
|
+
# Voice only, and stated plainly: a scenario written as though the person dialled in tests
|
|
509
|
+
# nothing when the agent is the one placing the call.
|
|
510
|
+
if self.modality == "voice" and self.call_direction:
|
|
511
|
+
parts.insert(
|
|
512
|
+
2,
|
|
513
|
+
f"CALL DIRECTION: {self.call_direction} - "
|
|
514
|
+
+ (
|
|
515
|
+
"this agent places the call, so the person did not dial and has no request "
|
|
516
|
+
"to open with"
|
|
517
|
+
if self.call_direction == "outbound"
|
|
518
|
+
else "people dial in to this agent"
|
|
519
|
+
),
|
|
520
|
+
)
|
|
521
|
+
if self.hard_constraints:
|
|
522
|
+
parts.append(
|
|
523
|
+
"HARD CONSTRAINTS the agent MUST follow (nothing may contradict these):\n - "
|
|
524
|
+
+ "\n - ".join(self.hard_constraints[:14])
|
|
525
|
+
)
|
|
526
|
+
if self.data_schema and full_schema:
|
|
527
|
+
parts.append(
|
|
528
|
+
"DATA SHAPE (the fields each record has):\n"
|
|
529
|
+
+ json.dumps(self.data_schema)[: 24000 if with_data else 2400]
|
|
530
|
+
)
|
|
531
|
+
if self.base_environment and with_data:
|
|
532
|
+
parts.append(
|
|
533
|
+
"THE AGENT'S REAL STARTING DATA. Reproduce this exactly, including anything\n"
|
|
534
|
+
"that looks like a mistake: a misspelled id, an item marked unavailable, an odd\n"
|
|
535
|
+
"price. The world is a replica of what the agent has, not a corrected version,\n"
|
|
536
|
+
"and a test written against a corrected world will not catch the real bug.\n"
|
|
537
|
+
+ json.dumps(self.base_environment, ensure_ascii=False)
|
|
538
|
+
)
|
|
539
|
+
if self.dependencies:
|
|
540
|
+
parts.append(
|
|
541
|
+
"WHAT THIS AGENT DEPENDS ON (the environment has to provide each of these):\n - "
|
|
542
|
+
+ "\n - ".join(
|
|
543
|
+
f"{one.name} ({one.kind or 'unspecified'}): {one.what}"
|
|
544
|
+
+ (f" — used by {', '.join(one.used_by)}" if one.used_by else "")
|
|
545
|
+
+ (f" — {_reached(one)}" if _reached(one) else "")
|
|
546
|
+
for one in self.dependencies
|
|
547
|
+
)
|
|
548
|
+
+ "\nThe agent's code is never edited; the environment must match its existing seam."
|
|
549
|
+
)
|
|
550
|
+
if self.real_use_cases:
|
|
551
|
+
# Every one of them. A scenario writer covers what it is shown, so a truncated
|
|
552
|
+
# list silently caps coverage at the cut rather than at the agent's surface.
|
|
553
|
+
parts.append(
|
|
554
|
+
"REAL USE CASES (what this agent is actually for):\n - "
|
|
555
|
+
+ "\n - ".join(self.real_use_cases)
|
|
556
|
+
)
|
|
557
|
+
if self.tool_entrypoints:
|
|
558
|
+
parts.append(
|
|
559
|
+
"THE AGENT'S OWN TOOL CODE. Run these rather than writing replacements:\n - "
|
|
560
|
+
+ "\n - ".join(
|
|
561
|
+
f"{one.tool}: {one.mode}"
|
|
562
|
+
+ (f" {one.module}.{one.callable}" if one.module else "")
|
|
563
|
+
+ (f", state passed as {one.first_arg}" if one.first_arg else "")
|
|
564
|
+
+ (f", build with {one.factory}" if one.factory else "")
|
|
565
|
+
for one in self.tool_entrypoints
|
|
566
|
+
)
|
|
567
|
+
)
|
|
568
|
+
if self.runtime_dependencies:
|
|
569
|
+
parts.append(
|
|
570
|
+
"RUNTIME CONNECTIONS (not business-world state; do not rebuild):\n"
|
|
571
|
+
+ "\n".join(
|
|
572
|
+
f" {one.name}: {one.what} {_reached(one)}"
|
|
573
|
+
for one in self.runtime_dependencies
|
|
574
|
+
)
|
|
575
|
+
)
|
|
576
|
+
if self.refusal_signature:
|
|
577
|
+
parts.append(
|
|
578
|
+
"HOW THIS AGENT REFUSES, in a value rather than by raising:\n "
|
|
579
|
+
f"{self.refusal_signature}"
|
|
580
|
+
)
|
|
581
|
+
if self.data_store:
|
|
582
|
+
store = self.data_store
|
|
583
|
+
parts.append(
|
|
584
|
+
"ITS DATA STORE:\n"
|
|
585
|
+
f" kind: {store.kind or 'unspecified'}\n"
|
|
586
|
+
f" connection comes from: {store.configured_by or 'unknown'}\n"
|
|
587
|
+
f" schema from: {store.schema_from or 'unknown'}\n"
|
|
588
|
+
f" its own loader: {store.loaded_by or 'none'}"
|
|
589
|
+
)
|
|
590
|
+
if self.runtime:
|
|
591
|
+
run = self.runtime
|
|
592
|
+
parts.append(
|
|
593
|
+
"RUNNING ITS CODE:\n"
|
|
594
|
+
f" {run.language or 'language unspecified'} {run.version}, "
|
|
595
|
+
f"install with {run.install or 'unknown'}"
|
|
596
|
+
+ (f", imports resolve from {run.workdir}" if run.workdir else "")
|
|
597
|
+
+ (
|
|
598
|
+
f", its own Compose file at {run.compose_file}"
|
|
599
|
+
if run.compose_file
|
|
600
|
+
else ""
|
|
601
|
+
)
|
|
602
|
+
+ (
|
|
603
|
+
f", its own Dockerfile at {run.dockerfile}"
|
|
604
|
+
if run.dockerfile
|
|
605
|
+
else ""
|
|
606
|
+
)
|
|
607
|
+
+ (
|
|
608
|
+
f", reached over {run.interface.kind} {run.interface.protocol} on "
|
|
609
|
+
f"port {run.interface.port}{run.interface.path}"
|
|
610
|
+
if run.interface
|
|
611
|
+
else ""
|
|
612
|
+
)
|
|
613
|
+
)
|
|
614
|
+
if self.notes:
|
|
615
|
+
parts.append(f"NOTES from reading the agent:\n{self.notes[:1500]}")
|
|
616
|
+
return "\n\n".join(parts)
|
|
617
|
+
|
|
618
|
+
def entry_for(self, tool: str) -> ToolEntry | None:
|
|
619
|
+
for one in self.tool_entrypoints:
|
|
620
|
+
# Coerced rather than assumed. Assigning this field directly bypasses validation, so
|
|
621
|
+
# an entry can arrive as a plain mapping, and reading it as an object would raise
|
|
622
|
+
# somewhere far from the assignment.
|
|
623
|
+
found = one if isinstance(one, ToolEntry) else ToolEntry(**dict(one))
|
|
624
|
+
if found.tool == tool:
|
|
625
|
+
return found
|
|
626
|
+
return None
|
|
627
|
+
|
|
628
|
+
def adoptable(self, tool: str) -> bool:
|
|
629
|
+
"""Whether this tool has code of its own that should be run instead of replaced."""
|
|
630
|
+
found = self.entry_for(tool)
|
|
631
|
+
return bool(found and found.mode in ("import", "construct", "service"))
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def is_data_free_conversation(contract: AgentContract) -> bool:
|
|
635
|
+
"""Whether the contract claims conversation without custom tools or business state.
|
|
636
|
+
|
|
637
|
+
Callers must also inspect the actual world; this claim alone is not a runtime exemption.
|
|
638
|
+
"""
|
|
639
|
+
store = (
|
|
640
|
+
contract.data_store.model_dump(exclude_defaults=True)
|
|
641
|
+
if contract.data_store
|
|
642
|
+
else {}
|
|
643
|
+
)
|
|
644
|
+
if store.get("kind") in {"", "none", "in_process"}:
|
|
645
|
+
store.pop("kind", None)
|
|
646
|
+
return bool(
|
|
647
|
+
contract.conversational
|
|
648
|
+
and not (
|
|
649
|
+
contract.tools
|
|
650
|
+
or contract.tool_entrypoints
|
|
651
|
+
or contract.dependencies
|
|
652
|
+
or contract.data_schema
|
|
653
|
+
or contract.base_environment
|
|
654
|
+
or store
|
|
655
|
+
)
|
|
656
|
+
)
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def validate_contract(contract: AgentContract) -> list[str]:
|
|
660
|
+
"""Structural problems that make a contract unusable downstream.
|
|
661
|
+
|
|
662
|
+
Deliberately narrow. This cannot tell whether the model read the agent correctly, only
|
|
663
|
+
whether the result is shaped well enough to build a world from. Semantic grounding is the
|
|
664
|
+
operator's job, which is why the harness surfaces the contract for review.
|
|
665
|
+
"""
|
|
666
|
+
problems: list[str] = []
|
|
667
|
+
if not contract.agent.strip():
|
|
668
|
+
problems.append("empty:agent")
|
|
669
|
+
for dependency in contract.dependencies:
|
|
670
|
+
if dependency.reached.dsn_env in {"LIVEKIT_URL", "LIVEKIT_INFERENCE_URL"}:
|
|
671
|
+
problems.append(
|
|
672
|
+
f"dependency[{dependency.name}]:runtime-connection-in-world — "
|
|
673
|
+
"LiveKit RTC/Inference is a runtime connection, not business-world data. "
|
|
674
|
+
"Move it to runtime_dependencies. Do not generate tables or tool sequences for it."
|
|
675
|
+
)
|
|
676
|
+
for dependency in contract.runtime_dependencies:
|
|
677
|
+
if (
|
|
678
|
+
dependency.kind.lower() in {"datastore", "database", "file", "queue"}
|
|
679
|
+
or dependency.reached.database
|
|
680
|
+
):
|
|
681
|
+
problems.append(
|
|
682
|
+
f"dependency[{dependency.name}]:business-data-in-runtime — "
|
|
683
|
+
"Business data belongs in dependencies, not runtime_dependencies."
|
|
684
|
+
)
|
|
685
|
+
for index, tool in enumerate(contract.tools):
|
|
686
|
+
if not tool.name.strip():
|
|
687
|
+
problems.append(f"tool[{index}]:no-name")
|
|
688
|
+
continue
|
|
689
|
+
unknown = sorted(set(tool.arg_types) - set(tool.args))
|
|
690
|
+
if unknown:
|
|
691
|
+
problems.append(
|
|
692
|
+
f"tool[{tool.name}]:types-for-unknown-args:{','.join(unknown)}"
|
|
693
|
+
)
|
|
694
|
+
# Conversational agents may expose no tools. Zero-argument tools are also legitimate;
|
|
695
|
+
# cardinality alone is not evidence that authoring omitted something.
|
|
696
|
+
if not contract.real_use_cases:
|
|
697
|
+
problems.append("no-use-cases")
|
|
698
|
+
# An `import`/`construct` entry without both halves is unusable: bundling compiles a binding
|
|
699
|
+
# from exactly these two fields. Caught here so the model is told while it can still fix the
|
|
700
|
+
# entry, rather than the run dying much later in `_compile_source_tool_handlers` after the
|
|
701
|
+
# runtime-validation attempts have been spent.
|
|
702
|
+
for entry in contract.tool_entrypoints:
|
|
703
|
+
if entry.mode not in {"import", "construct"}:
|
|
704
|
+
continue
|
|
705
|
+
if entry.module.strip() and entry.callable.strip():
|
|
706
|
+
continue
|
|
707
|
+
problems.append(
|
|
708
|
+
f"tool_entrypoint[{entry.tool}]:{entry.mode}-needs-module-and-callable — "
|
|
709
|
+
"give the importable module path and the callable name, or record a mode that "
|
|
710
|
+
"matches what the repository actually exposes."
|
|
711
|
+
)
|
|
712
|
+
# Iterate the tools, not tool_names(): that returns a set, so duplicates collapse before
|
|
713
|
+
# they can be counted and the check silently never fires.
|
|
714
|
+
names = [tool.name for tool in contract.tools if tool.name.strip()]
|
|
715
|
+
duplicates = sorted({name for name in names if names.count(name) > 1})
|
|
716
|
+
if duplicates:
|
|
717
|
+
problems.append(f"duplicate-tool-names:{','.join(duplicates)}")
|
|
718
|
+
return problems
|