agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,413 @@
|
|
|
1
|
+
"""Whether a generated world is usable, decided by exercising it.
|
|
2
|
+
|
|
3
|
+
Published work on synthesised environments is consistent about two things. Most generated
|
|
4
|
+
environments contain bugs, so the gate has to aim at the ones that block rather than at
|
|
5
|
+
perfection. And the bugs cluster: edge-case handling first, then state consistency across
|
|
6
|
+
several calls. A gate that runs each handler once and calls it done misses both clusters.
|
|
7
|
+
|
|
8
|
+
So this exercises every tool three ways, and then exercises the world as a sequence:
|
|
9
|
+
|
|
10
|
+
- **happy**: a valid call, built from the values the contract says the argument accepts
|
|
11
|
+
- **edge**: an identifier that does not exist, and a required argument left out
|
|
12
|
+
- **sequence**: a declared series of calls whose final state is asserted
|
|
13
|
+
|
|
14
|
+
The distinction that matters throughout is **refusal versus crash**. A tool that rejects a
|
|
15
|
+
nonexistent id is working: that refusal is the entire point of a real world. A tool that raises
|
|
16
|
+
``KeyError`` on the same input is broken. They are both failures to a naive check and opposite
|
|
17
|
+
outcomes here.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import re
|
|
23
|
+
from dataclasses import dataclass, field
|
|
24
|
+
from typing import Any, Iterable, Mapping, Sequence
|
|
25
|
+
|
|
26
|
+
from ..contract import AgentContract, ToolSpec
|
|
27
|
+
from .expectations import check_state
|
|
28
|
+
from .kinds import WorldKind, for_contract
|
|
29
|
+
from .kinds import resolve as _resolve_kind
|
|
30
|
+
from .runtime import GeneratedWorld
|
|
31
|
+
|
|
32
|
+
HAPPY = "happy"
|
|
33
|
+
EDGE = "edge"
|
|
34
|
+
SEQUENCE = "sequence"
|
|
35
|
+
COVERAGE = "coverage"
|
|
36
|
+
DATA = "data"
|
|
37
|
+
|
|
38
|
+
# A value no generated world should ever have seeded, used to prove a lookup refuses.
|
|
39
|
+
ABSENT = "__does_not_exist__"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class ProbeResult:
|
|
44
|
+
name: str
|
|
45
|
+
kind: str
|
|
46
|
+
passed: bool
|
|
47
|
+
detail: str = ""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class ProbeReport:
|
|
52
|
+
results: list[ProbeResult] = field(default_factory=list)
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def score(self) -> float:
|
|
56
|
+
return (
|
|
57
|
+
sum(1 for result in self.results if result.passed) / len(self.results)
|
|
58
|
+
if self.results
|
|
59
|
+
else 0.0
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def failures(self) -> list[ProbeResult]:
|
|
64
|
+
return [result for result in self.results if not result.passed]
|
|
65
|
+
|
|
66
|
+
def summary(self) -> str:
|
|
67
|
+
if not self.results:
|
|
68
|
+
return "no probes ran"
|
|
69
|
+
lines = [
|
|
70
|
+
f"{len(self.results) - len(self.failures)}/{len(self.results)} probes passed"
|
|
71
|
+
]
|
|
72
|
+
for failure in self.failures:
|
|
73
|
+
lines.append(f" {failure.kind}:{failure.name}: {failure.detail}")
|
|
74
|
+
return "\n".join(lines)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _valid_arguments(tool: ToolSpec) -> dict[str, Any]:
|
|
78
|
+
"""A plausible call, using the values the contract says each argument accepts."""
|
|
79
|
+
arguments: dict[str, Any] = {}
|
|
80
|
+
for arg in tool.args:
|
|
81
|
+
options = tool.arg_values.get(arg)
|
|
82
|
+
if isinstance(options, (list, tuple)):
|
|
83
|
+
usable = [value for value in options if value not in (None, "null", "")]
|
|
84
|
+
if usable:
|
|
85
|
+
arguments[arg] = usable[0]
|
|
86
|
+
continue
|
|
87
|
+
declared = tool.arg_types.get(arg, "")
|
|
88
|
+
if "list" in declared:
|
|
89
|
+
arguments[arg] = []
|
|
90
|
+
elif "int" in declared:
|
|
91
|
+
arguments[arg] = 1
|
|
92
|
+
elif "bool" in declared:
|
|
93
|
+
arguments[arg] = True
|
|
94
|
+
else:
|
|
95
|
+
arguments[arg] = ABSENT
|
|
96
|
+
return arguments
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _is_a_real_identifier(value: Any) -> bool:
|
|
100
|
+
"""Whether a permitted value names a record, rather than being an enum like 'M' or 'null'."""
|
|
101
|
+
if not isinstance(value, str) or value in ("", "null", "none", "None"):
|
|
102
|
+
return False
|
|
103
|
+
return len(value) > 2 and not value.isdigit()
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _missing_catalogue(
|
|
107
|
+
world: GeneratedWorld, contract: AgentContract, kind: WorldKind
|
|
108
|
+
) -> list[str]:
|
|
109
|
+
"""Identifiers the contract says a tool accepts that are nowhere in the seeded world.
|
|
110
|
+
|
|
111
|
+
The gap this closes is a whole category left unseeded. Every call naming a sauce then fails,
|
|
112
|
+
which looks from the outside exactly like a world being correctly strict, and a suite where
|
|
113
|
+
nothing can be ordered scores perfectly. Whether the catalogue is complete cannot be settled
|
|
114
|
+
by behaviour, so it is checked against the data.
|
|
115
|
+
"""
|
|
116
|
+
present = kind.values_present(world)
|
|
117
|
+
missing: list[str] = []
|
|
118
|
+
for tool in contract.tools:
|
|
119
|
+
for arg, values in (tool.arg_values or {}).items():
|
|
120
|
+
if not isinstance(values, (list, tuple)):
|
|
121
|
+
continue
|
|
122
|
+
if not _looks_like_an_identifier(arg, tool.arg_types.get(arg, "")):
|
|
123
|
+
continue
|
|
124
|
+
absent = [
|
|
125
|
+
value
|
|
126
|
+
for value in values
|
|
127
|
+
if _is_a_real_identifier(value) and value not in present
|
|
128
|
+
]
|
|
129
|
+
if absent:
|
|
130
|
+
shown = ", ".join(absent[:4]) + (
|
|
131
|
+
f" and {len(absent) - 4} more" if len(absent) > 4 else ""
|
|
132
|
+
)
|
|
133
|
+
missing.append(f"{tool.name}.{arg}: {shown}")
|
|
134
|
+
return missing
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _missing_argument(error: str) -> bool:
|
|
138
|
+
"""Whether a failure is the language rejecting a call for want of a required argument."""
|
|
139
|
+
said = (error or "").lower()
|
|
140
|
+
return (
|
|
141
|
+
"typeerror" in said
|
|
142
|
+
and "argument" in said
|
|
143
|
+
and ("missing" in said or "required" in said or "unexpected keyword" in said)
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _reads_argument(source: str, name: str) -> bool:
|
|
148
|
+
"""Whether a handler actually takes this argument out of ``args``.
|
|
149
|
+
|
|
150
|
+
Looking for the bare name is not enough. A handler that reads ``args['order_ids']`` and then
|
|
151
|
+
loops ``for order_id in order_ids`` mentions ``order_id`` all over itself while never reading
|
|
152
|
+
the argument the tool is given, so it silently ignores its input and reports success or
|
|
153
|
+
refuses everything. Both look fine from the outside, which is why this is checked at the
|
|
154
|
+
point of access rather than by behaviour.
|
|
155
|
+
"""
|
|
156
|
+
pattern = (
|
|
157
|
+
rf"args\s*(?:\[\s*|\.get\s*\(\s*|\.pop\s*\(\s*)"
|
|
158
|
+
rf"['\"]{re.escape(name)}['\"]"
|
|
159
|
+
)
|
|
160
|
+
return re.search(pattern, source) is not None
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _looks_like_an_identifier(name: str, _declared: str = "") -> bool:
|
|
164
|
+
"""Whether an argument names a record that has to exist for the call to make sense.
|
|
165
|
+
|
|
166
|
+
Decided by the name alone. Treating every ``str`` argument as a catalogue was a trap: a
|
|
167
|
+
``size`` accepting "Medium" and "Large" then demanded rows called Medium and Large in the
|
|
168
|
+
world, which can never be seeded sensibly. The only ways out were to invent nonsense rows or
|
|
169
|
+
to edit the contract, so a check meant to catch a missing menu instead pushed towards
|
|
170
|
+
corrupting the record of what the agent is.
|
|
171
|
+
|
|
172
|
+
A missed catalogue is a check that does not fire. A false one is a stage with no legal move,
|
|
173
|
+
which is much worse, so this stays narrow.
|
|
174
|
+
"""
|
|
175
|
+
return (
|
|
176
|
+
name.endswith(("_id", "_ids", "_key", "_ref", "_code", "_sku")) or name == "id"
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _identifier_arguments(tool: ToolSpec) -> dict[str, Any] | None:
|
|
181
|
+
"""The same call with every identifier replaced by one that cannot exist.
|
|
182
|
+
|
|
183
|
+
Deliberately not gated on the contract listing that argument's values. A contract that
|
|
184
|
+
failed to record them is exactly the case where nobody has checked what this tool does with
|
|
185
|
+
a bad id, so skipping the probe there drops it precisely where it is most needed.
|
|
186
|
+
"""
|
|
187
|
+
arguments = _valid_arguments(tool)
|
|
188
|
+
swapped = False
|
|
189
|
+
for arg in tool.args:
|
|
190
|
+
declared = tool.arg_types.get(arg, "")
|
|
191
|
+
if not tool.arg_values.get(arg) and not _looks_like_an_identifier(
|
|
192
|
+
arg, declared
|
|
193
|
+
):
|
|
194
|
+
continue
|
|
195
|
+
arguments[arg] = [ABSENT] if "list" in declared.lower() else ABSENT
|
|
196
|
+
swapped = True
|
|
197
|
+
return arguments if swapped else None
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def probe(
|
|
201
|
+
world: GeneratedWorld,
|
|
202
|
+
contract: AgentContract,
|
|
203
|
+
*,
|
|
204
|
+
sequences: Iterable[Mapping[str, Any]] = (),
|
|
205
|
+
kind: WorldKind | None = None,
|
|
206
|
+
) -> ProbeReport:
|
|
207
|
+
"""Exercise the world and report what it can and cannot do.
|
|
208
|
+
|
|
209
|
+
``sequences`` are declared by whoever built the world, because knowing that adding an item
|
|
210
|
+
should make it appear in a listing is judgement about this agent, not something derivable
|
|
211
|
+
from a schema.
|
|
212
|
+
"""
|
|
213
|
+
report = ProbeReport()
|
|
214
|
+
kind = kind or for_contract(contract)
|
|
215
|
+
|
|
216
|
+
# Every probe runs from the same starting world. Probes mutate, so without reverting
|
|
217
|
+
# between them each one inherits the debris of the last and a check expecting three rows
|
|
218
|
+
# finds seven. That is a fault in the harness, not in the world being checked.
|
|
219
|
+
baseline = world.checkpoint()
|
|
220
|
+
runtime_tools = set(getattr(world, "runtime_tools", set()))
|
|
221
|
+
|
|
222
|
+
for tool in contract.tools:
|
|
223
|
+
if tool.name not in world.handlers and tool.name not in runtime_tools:
|
|
224
|
+
report.results.append(
|
|
225
|
+
ProbeResult(tool.name, COVERAGE, False, "contract tool has no handler")
|
|
226
|
+
)
|
|
227
|
+
for name in world.handlers:
|
|
228
|
+
if name not in contract.tool_names():
|
|
229
|
+
report.results.append(
|
|
230
|
+
ProbeResult(
|
|
231
|
+
name, COVERAGE, False, "handler for a tool the agent does not have"
|
|
232
|
+
)
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
for gap in _missing_catalogue(world, contract, kind):
|
|
236
|
+
report.results.append(
|
|
237
|
+
ProbeResult(
|
|
238
|
+
gap.split(":")[0],
|
|
239
|
+
DATA,
|
|
240
|
+
False,
|
|
241
|
+
f"the contract accepts values the world does not have: {gap}",
|
|
242
|
+
)
|
|
243
|
+
)
|
|
244
|
+
if not _missing_catalogue(world, contract, kind):
|
|
245
|
+
report.results.append(
|
|
246
|
+
ProbeResult("catalogue", DATA, True, "every permitted identifier exists")
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
for tool in contract.tools:
|
|
250
|
+
if tool.name in runtime_tools:
|
|
251
|
+
report.results.append(
|
|
252
|
+
ProbeResult(
|
|
253
|
+
tool.name,
|
|
254
|
+
COVERAGE,
|
|
255
|
+
True,
|
|
256
|
+
"executes inside the submitted agent runtime",
|
|
257
|
+
)
|
|
258
|
+
)
|
|
259
|
+
continue
|
|
260
|
+
if tool.name not in world.handlers:
|
|
261
|
+
continue
|
|
262
|
+
source = world.handlers[tool.name]
|
|
263
|
+
# Reading the source only says anything about a handler written here. A tool bound to the
|
|
264
|
+
# agent's own code has a handler that forwards every argument on, so it never names any of
|
|
265
|
+
# them, and checking for the names would fail every adopted tool while telling nobody
|
|
266
|
+
# anything. The names are the agent's own problem there, and its own code is what runs.
|
|
267
|
+
if not contract.adoptable(tool.name):
|
|
268
|
+
unread = [arg for arg in tool.args if not _reads_argument(source, arg)]
|
|
269
|
+
report.results.append(
|
|
270
|
+
ProbeResult(
|
|
271
|
+
tool.name,
|
|
272
|
+
COVERAGE,
|
|
273
|
+
not unread,
|
|
274
|
+
# A handler reading order_ids when the tool takes order_id refuses
|
|
275
|
+
# everything, which looks exactly like a handler correctly refusing a bad id.
|
|
276
|
+
# Behaviour alone cannot tell those apart, so the names are checked directly.
|
|
277
|
+
""
|
|
278
|
+
if not unread
|
|
279
|
+
else f"never reads {', '.join(unread)}, which the contract says it takes",
|
|
280
|
+
)
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
world.revert(baseline)
|
|
284
|
+
call = world.call(tool.name, _valid_arguments(tool))
|
|
285
|
+
# A refusal here is acceptable: the contract's first listed value may genuinely be
|
|
286
|
+
# invalid in the seeded world. A crash never is.
|
|
287
|
+
report.results.append(
|
|
288
|
+
ProbeResult(
|
|
289
|
+
tool.name,
|
|
290
|
+
HAPPY,
|
|
291
|
+
call.ok or call.refused,
|
|
292
|
+
"" if call.ok or call.refused else call.error,
|
|
293
|
+
)
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
bogus = _identifier_arguments(tool)
|
|
297
|
+
if bogus is not None:
|
|
298
|
+
world.revert(baseline)
|
|
299
|
+
call = world.call(tool.name, bogus)
|
|
300
|
+
report.results.append(
|
|
301
|
+
ProbeResult(
|
|
302
|
+
tool.name,
|
|
303
|
+
EDGE,
|
|
304
|
+
call.refused,
|
|
305
|
+
""
|
|
306
|
+
if call.refused
|
|
307
|
+
else (
|
|
308
|
+
"succeeded on an id that does not exist"
|
|
309
|
+
if call.ok
|
|
310
|
+
else f"crashed instead of refusing: {call.error}"
|
|
311
|
+
),
|
|
312
|
+
)
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
if tool.args:
|
|
316
|
+
world.revert(baseline)
|
|
317
|
+
missing = _valid_arguments(tool)
|
|
318
|
+
missing.pop(tool.args[0], None)
|
|
319
|
+
call = world.call(tool.name, missing)
|
|
320
|
+
# A tool bound to the agent's own code is a function with real parameters, so leaving a
|
|
321
|
+
# required one out is rejected by the language before the body runs. That is the call
|
|
322
|
+
# being refused, not the world falling over, and counting it as a crash would fail
|
|
323
|
+
# every adopted tool for behaving exactly as the agent's own runtime makes it behave.
|
|
324
|
+
declined = call.refused or (
|
|
325
|
+
contract.adoptable(tool.name) and _missing_argument(call.error)
|
|
326
|
+
)
|
|
327
|
+
report.results.append(
|
|
328
|
+
ProbeResult(
|
|
329
|
+
f"{tool.name}:without-{tool.args[0]}",
|
|
330
|
+
EDGE,
|
|
331
|
+
declined,
|
|
332
|
+
""
|
|
333
|
+
if declined
|
|
334
|
+
else (
|
|
335
|
+
"accepted a call with a required argument missing"
|
|
336
|
+
if call.ok
|
|
337
|
+
else f"crashed instead of refusing: {call.error}"
|
|
338
|
+
),
|
|
339
|
+
)
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
world.revert(baseline)
|
|
343
|
+
unknown = world.call(ABSENT, {})
|
|
344
|
+
report.results.append(
|
|
345
|
+
ProbeResult(
|
|
346
|
+
"unknown-tool",
|
|
347
|
+
EDGE,
|
|
348
|
+
unknown.refused,
|
|
349
|
+
"" if unknown.refused else "an unknown tool did not refuse",
|
|
350
|
+
)
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
for index, sequence in enumerate(sequences):
|
|
354
|
+
world.revert(baseline)
|
|
355
|
+
report.results.append(_run_sequence(world, sequence, index))
|
|
356
|
+
|
|
357
|
+
# Leave the world as the builder left it, not as the last probe left it.
|
|
358
|
+
world.revert(baseline)
|
|
359
|
+
return report
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def dirty_state(
|
|
363
|
+
world: GeneratedWorld,
|
|
364
|
+
sequences: Iterable[Mapping[str, Any]],
|
|
365
|
+
kind: WorldKind | None = None,
|
|
366
|
+
) -> list[str]:
|
|
367
|
+
"""Tables a scenario writes to that already hold rows before anything has happened.
|
|
368
|
+
|
|
369
|
+
A world is the state every scenario starts from, so an order table with rows in it means
|
|
370
|
+
the builder's own testing was frozen into the base state. Every scenario then begins with
|
|
371
|
+
somebody else's order already in the cart, and a count check that should read one reads
|
|
372
|
+
seven. Which tables are transactional is not guessable from a schema, so it is worked out
|
|
373
|
+
by running the declared sequences and seeing what moves.
|
|
374
|
+
"""
|
|
375
|
+
kind = kind or _resolve_kind("sqlite")
|
|
376
|
+
baseline = world.checkpoint()
|
|
377
|
+
before = kind.mutable_state(world)
|
|
378
|
+
touched: set[str] = set()
|
|
379
|
+
for index, sequence in enumerate(sequences):
|
|
380
|
+
world.revert(baseline)
|
|
381
|
+
_run_sequence(world, sequence, index)
|
|
382
|
+
for name, size in kind.mutable_state(world).items():
|
|
383
|
+
if size != before.get(name, 0):
|
|
384
|
+
touched.add(name)
|
|
385
|
+
world.revert(baseline)
|
|
386
|
+
return sorted(name for name in touched if before.get(name, 0) > 0)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _run_sequence(
|
|
390
|
+
world: GeneratedWorld, sequence: Mapping[str, Any], index: int
|
|
391
|
+
) -> ProbeResult:
|
|
392
|
+
"""Run a declared series of calls and check the state it leaves behind.
|
|
393
|
+
|
|
394
|
+
This is the state-consistency check: the failure mode where each call works on its own and
|
|
395
|
+
the world still forgets what the previous one did.
|
|
396
|
+
"""
|
|
397
|
+
name = str(sequence.get("name") or f"sequence-{index}")
|
|
398
|
+
calls: Sequence[Mapping[str, Any]] = sequence.get("calls") or ()
|
|
399
|
+
for step in calls:
|
|
400
|
+
call = world.call(str(step.get("tool", "")), step.get("arguments") or {})
|
|
401
|
+
if step.get("expect") == "refusal":
|
|
402
|
+
if not call.refused:
|
|
403
|
+
return ProbeResult(
|
|
404
|
+
name, SEQUENCE, False, f"{call.name} should have refused"
|
|
405
|
+
)
|
|
406
|
+
continue
|
|
407
|
+
if not call.ok:
|
|
408
|
+
return ProbeResult(name, SEQUENCE, False, f"{call.name}: {call.error}")
|
|
409
|
+
|
|
410
|
+
failures = check_state(world.state(), sequence.get("expect_state") or {})
|
|
411
|
+
if failures:
|
|
412
|
+
return ProbeResult(name, SEQUENCE, False, failures[0])
|
|
413
|
+
return ProbeResult(name, SEQUENCE, True)
|