agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/harness/prove.py
ADDED
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""Proving a scenario is worth keeping, before anything is ever run against the agent.
|
|
2
|
+
|
|
3
|
+
Three gates, all pure code. No model is asked whether a scenario is good; the environment
|
|
4
|
+
decides. Terminal-bench keeps its tasks honest this way, and it is the cheapest useful thing in
|
|
5
|
+
the whole harness: no tokens, no network, a few milliseconds.
|
|
6
|
+
|
|
7
|
+
**Ready.** Reset the world, run the scenario's own ``setup.py``, then its ``ready.py``. The world
|
|
8
|
+
has to hold what the scenario presumes. A scenario about the last five chocolates is only a test
|
|
9
|
+
of the agent if there really are five; otherwise the agent fails for something we got wrong and
|
|
10
|
+
it reads as the agent's fault. This gate is why a missing precondition can never be mistaken for
|
|
11
|
+
a finding.
|
|
12
|
+
|
|
13
|
+
**Solvable.** Then run the reference solution and the checks. They must pass. If they do not,
|
|
14
|
+
either the scenario cannot be passed at all or its checks are wrong, and both have happened
|
|
15
|
+
here: one scenario asserted a value the agent was never permitted to send; another demanded
|
|
16
|
+
confirmation of an item that could not be ordered. Neither was noticed until a live run failed
|
|
17
|
+
and read as a finding about the agent.
|
|
18
|
+
|
|
19
|
+
**Not vacuous.** Then reset, set up again, run *nothing*, and run the checks. They must fail. A
|
|
20
|
+
check that passes with no actions taken grades nothing while reporting a result, which is how a
|
|
21
|
+
suite goes quietly green. This one earns its keep: on a third-party benchmark it caught three
|
|
22
|
+
sub-goals that passed trivially because the seeded world already contained a cancelled order.
|
|
23
|
+
|
|
24
|
+
Only a scenario that clears all three is kept. That is the green light.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import logging
|
|
30
|
+
from dataclasses import dataclass, field, replace
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
from .catalogue import Catalogue
|
|
34
|
+
from .checks import Outcome, run_check
|
|
35
|
+
from .folder import apply_setup, check_ready
|
|
36
|
+
from .scenario import Scenario
|
|
37
|
+
from .world.runtime import Call, GeneratedWorld
|
|
38
|
+
from .world.snapshot import restore
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class Proof:
|
|
43
|
+
"""Whether a scenario holds up, and what happened when it was tried."""
|
|
44
|
+
|
|
45
|
+
ready: bool = False
|
|
46
|
+
solvable: bool = False
|
|
47
|
+
vacuous: bool = True
|
|
48
|
+
why_not_ready: str = ""
|
|
49
|
+
# Checks that held with nothing done. The scenario is only vacuous when *every* check does
|
|
50
|
+
# that, but a single one still grades nothing, and since sub-goals are shared it will report
|
|
51
|
+
# itself as held for an agent that did nothing at all. Named rather than refused: on a
|
|
52
|
+
# scenario about a refusal, "no order was placed" holding on an untouched world is correct.
|
|
53
|
+
weak: list[str] = field(default_factory=list)
|
|
54
|
+
failed_attempts: list[str] = field(default_factory=list)
|
|
55
|
+
with_solution: list[Outcome] = field(default_factory=list)
|
|
56
|
+
with_nothing: list[Outcome] = field(default_factory=list)
|
|
57
|
+
refused: list[str] = field(default_factory=list)
|
|
58
|
+
broken: list[str] = field(default_factory=list)
|
|
59
|
+
# Solution steps that were recorded without being executed, because the tool they name has no
|
|
60
|
+
# endpoint bound in this lane. The condition was already detected and logged and then went
|
|
61
|
+
# nowhere, so a scenario whose entire solution was assumed saved as "proved". Carried here so
|
|
62
|
+
# whoever keeps the scenario is told what the proof did not cover.
|
|
63
|
+
assumed: list[str] = field(default_factory=list)
|
|
64
|
+
# Graded by judgement alone, because the world holds nothing a check could read. True only for
|
|
65
|
+
# a target we cannot see into, never a way around writing a check for a world that has one.
|
|
66
|
+
judged_only: bool = False
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def holds(self) -> bool:
|
|
70
|
+
return (
|
|
71
|
+
self.ready
|
|
72
|
+
and self.solvable
|
|
73
|
+
and not self.vacuous
|
|
74
|
+
and not self.weak
|
|
75
|
+
and not self.failed_attempts
|
|
76
|
+
and not self.broken
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
def gates(self) -> dict[str, bool]:
|
|
80
|
+
"""The three answers, for anything that wants to show them."""
|
|
81
|
+
return {
|
|
82
|
+
"ready": self.ready,
|
|
83
|
+
"solvable": self.solvable,
|
|
84
|
+
"not_vacuous": not self.vacuous and not self.failed_attempts,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
def why(self) -> str:
|
|
88
|
+
"""What to fix, in the order worth fixing it."""
|
|
89
|
+
if not self.ready:
|
|
90
|
+
return (
|
|
91
|
+
"the world is not ready for this scenario, so running it would test us rather "
|
|
92
|
+
f"than the agent:\n - {self.why_not_ready}\n\n"
|
|
93
|
+
"Either setup.py does not make the change this scenario needs, or ready.py is "
|
|
94
|
+
"checking for something the setup never creates."
|
|
95
|
+
)
|
|
96
|
+
if self.broken:
|
|
97
|
+
return "these checks are broken, not failing:\n - " + "\n - ".join(
|
|
98
|
+
self.broken
|
|
99
|
+
)
|
|
100
|
+
if not self.solvable:
|
|
101
|
+
failed = [one for one in self.with_solution if not one.held]
|
|
102
|
+
said = "\n - ".join(f"{one.name}: {one.said}" for one in failed)
|
|
103
|
+
refusals = (
|
|
104
|
+
"\n\nThe solution's own calls were refused by the world:\n - "
|
|
105
|
+
+ "\n - ".join(self.refused)
|
|
106
|
+
if self.refused
|
|
107
|
+
else ""
|
|
108
|
+
)
|
|
109
|
+
return (
|
|
110
|
+
"the reference solution does not pass this scenario's own checks, so either the "
|
|
111
|
+
"scenario cannot be passed or the checks are wrong:\n - "
|
|
112
|
+
+ said
|
|
113
|
+
+ refusals
|
|
114
|
+
)
|
|
115
|
+
if self.vacuous:
|
|
116
|
+
passed = [one.name for one in self.with_nothing if one.held]
|
|
117
|
+
return (
|
|
118
|
+
"these checks pass without the agent doing anything, so they grade nothing:\n - "
|
|
119
|
+
+ "\n - ".join(passed)
|
|
120
|
+
+ "\n\nIf the point of this scenario is that nothing should happen, checking "
|
|
121
|
+
"the world alone cannot show it — an untouched world looks identical to one "
|
|
122
|
+
"where the agent did nothing at all. Check the calls instead: that the agent "
|
|
123
|
+
"tried, and that the attempt was refused rather than succeeding.\n"
|
|
124
|
+
" def check(world, calls):\n"
|
|
125
|
+
" tried = [c for c in calls if c.name == 'add']\n"
|
|
126
|
+
" if not tried: return 'never attempted it'\n"
|
|
127
|
+
" if any(c.ok for c in tried): return 'it succeeded'\n"
|
|
128
|
+
" return None"
|
|
129
|
+
)
|
|
130
|
+
if self.failed_attempts:
|
|
131
|
+
return (
|
|
132
|
+
"these checks award credit when every tool attempt crashes without changing "
|
|
133
|
+
"the world or returning a result:\n - "
|
|
134
|
+
+ "\n - ".join(self.failed_attempts)
|
|
135
|
+
+ "\n\nArguments and a tool name are not evidence of success. Require a "
|
|
136
|
+
"successful outcome or the corresponding world effect. For a refusal, require "
|
|
137
|
+
"evidence of the intended refusal, not an arbitrary execution error."
|
|
138
|
+
)
|
|
139
|
+
if self.weak:
|
|
140
|
+
return (
|
|
141
|
+
"these individual checks pass without the agent doing anything, so keeping them "
|
|
142
|
+
"would display credit unsupported by evidence:\n - "
|
|
143
|
+
+ "\n - ".join(self.weak)
|
|
144
|
+
+ "\n\nMake each check assert the relevant attempt or observation as well as the "
|
|
145
|
+
"final state. For a refusal, require that the call was attempted and refused; "
|
|
146
|
+
"an untouched world is not evidence that the agent refused correctly."
|
|
147
|
+
)
|
|
148
|
+
return "holds"
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _checks_for(scenario: Scenario, catalogue: Catalogue) -> list[tuple[str, str]]:
|
|
152
|
+
"""The deterministic checks this scenario is graded by, in catalogue order."""
|
|
153
|
+
chosen: list[tuple[str, str]] = []
|
|
154
|
+
for name in scenario.sub_goals:
|
|
155
|
+
sub_goal = catalogue.named(name)
|
|
156
|
+
if sub_goal is not None and sub_goal.deterministic():
|
|
157
|
+
chosen.append((name, sub_goal.check))
|
|
158
|
+
return chosen
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _something_to_check(world_root: Path) -> bool:
|
|
162
|
+
"""Whether this world holds anything a check in code could read.
|
|
163
|
+
|
|
164
|
+
A world with collections, or a tool with an endpoint bound, can be inspected, so a scenario
|
|
165
|
+
graded only by a model is a scenario that chose not to look. A world with neither cannot be
|
|
166
|
+
inspected at all, and insisting on a code check there produces checks that assert on nothing.
|
|
167
|
+
Unreadable is treated as "there is something", because refusing is the safe direction.
|
|
168
|
+
"""
|
|
169
|
+
try:
|
|
170
|
+
world = restore(world_root)
|
|
171
|
+
except Exception: # noqa: BLE001 - a world we cannot open is not a world we can excuse
|
|
172
|
+
return True
|
|
173
|
+
try:
|
|
174
|
+
has_rows = any(rows for rows in world.state().values())
|
|
175
|
+
endpoints = getattr(world, "endpoint_for", {}) or {}
|
|
176
|
+
return bool(has_rows or any(endpoints.values()))
|
|
177
|
+
except Exception: # noqa: BLE001 - same reasoning
|
|
178
|
+
return True
|
|
179
|
+
finally:
|
|
180
|
+
world.close()
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def prepared(
|
|
184
|
+
scenario: Scenario, world_root: Path
|
|
185
|
+
) -> tuple[GeneratedWorld, Outcome, Outcome]:
|
|
186
|
+
"""A fresh world with this scenario's setup applied, and how that went."""
|
|
187
|
+
world = restore(world_root)
|
|
188
|
+
world.reset()
|
|
189
|
+
applied = apply_setup(scenario, world)
|
|
190
|
+
ready = check_ready(scenario, world) if applied.ok else Outcome(False, applied.said)
|
|
191
|
+
# The setup's own calls are not the agent's. Clearing them keeps a check that counts calls
|
|
192
|
+
# from crediting the agent with work the scenario did on its behalf.
|
|
193
|
+
world.calls = []
|
|
194
|
+
return world, applied, ready
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
logger = logging.getLogger(__name__)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def play_reference_step(world: GeneratedWorld, step: object) -> Call:
|
|
201
|
+
"""Play one correct-agent step without confusing its API with its dependency's API.
|
|
202
|
+
|
|
203
|
+
A generated world can execute the agent-facing call directly. A source-provisioned world
|
|
204
|
+
cannot: its model-facing tool may enforce local ordering and inject session state before it
|
|
205
|
+
reaches a raw HTTP service. For those worlds, drive the raw endpoint with the explicitly
|
|
206
|
+
declared environment payload, but record only the semantic call the agent was expected to
|
|
207
|
+
make. Purely local state-machine tools have no dependency effect and are recorded as such.
|
|
208
|
+
"""
|
|
209
|
+
name = str(getattr(step, "tool", "") or "")
|
|
210
|
+
arguments = dict(getattr(step, "arguments", {}) or {})
|
|
211
|
+
environment_arguments = _resolve_reference_values(
|
|
212
|
+
dict(getattr(step, "environment_arguments", {}) or {}), world.calls
|
|
213
|
+
)
|
|
214
|
+
runtime_tools = set(getattr(world, "runtime_tools", set()))
|
|
215
|
+
if name not in runtime_tools:
|
|
216
|
+
return world.call(name, arguments)
|
|
217
|
+
|
|
218
|
+
endpoint = getattr(world, "endpoint_for", {}).get(name)
|
|
219
|
+
forward = getattr(world, "forward", None)
|
|
220
|
+
if endpoint and callable(forward):
|
|
221
|
+
dependency_arguments = environment_arguments or arguments
|
|
222
|
+
effect = forward(endpoint, dependency_arguments, record=False)
|
|
223
|
+
semantic = Call(
|
|
224
|
+
name=name,
|
|
225
|
+
arguments=arguments,
|
|
226
|
+
result=effect.result,
|
|
227
|
+
ok=effect.ok,
|
|
228
|
+
error=effect.error,
|
|
229
|
+
refused=effect.refused,
|
|
230
|
+
at=effect.at,
|
|
231
|
+
)
|
|
232
|
+
else:
|
|
233
|
+
# Reaching here means the tool is a declared runtime tool with nothing bound to call.
|
|
234
|
+
# Two very different situations arrive at the same place: a local orchestration action
|
|
235
|
+
# that genuinely has no environment effect, and a lane where the runtime is built after
|
|
236
|
+
# authoring so no endpoint exists yet. The second cannot be proved, and recording it as
|
|
237
|
+
# a pass is what lets a scenario be kept on a solution nothing ever executed. Say so.
|
|
238
|
+
logger.warning(
|
|
239
|
+
"reference step assumed rather than executed: tool=%s reason=%s. Its result is "
|
|
240
|
+
"recorded ok with no call made, so this step proves nothing about the world.",
|
|
241
|
+
name,
|
|
242
|
+
"no endpoint bound" if not endpoint else "no forwarder on the world",
|
|
243
|
+
)
|
|
244
|
+
semantic = Call(name=name, arguments=arguments)
|
|
245
|
+
world.calls.append(semantic)
|
|
246
|
+
return semantic
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _resolve_reference_values(value: object, calls: list[Call]) -> object:
|
|
250
|
+
"""Resolve a dependency value produced by an earlier reference call.
|
|
251
|
+
|
|
252
|
+
``$call.book_ride.booking_ref`` refers to the named field on the most recent successful
|
|
253
|
+
``book_ride`` result. This drives real chained effects without pretending the model knew a
|
|
254
|
+
backend-generated id.
|
|
255
|
+
"""
|
|
256
|
+
if isinstance(value, dict):
|
|
257
|
+
return {
|
|
258
|
+
key: _resolve_reference_values(item, calls) for key, item in value.items()
|
|
259
|
+
}
|
|
260
|
+
if isinstance(value, list):
|
|
261
|
+
return [_resolve_reference_values(item, calls) for item in value]
|
|
262
|
+
if not isinstance(value, str) or not value.startswith("$call."):
|
|
263
|
+
return value
|
|
264
|
+
parts = value.split(".")
|
|
265
|
+
if len(parts) < 3:
|
|
266
|
+
return value
|
|
267
|
+
call_name = parts[1]
|
|
268
|
+
found = next(
|
|
269
|
+
(call for call in reversed(calls) if call.name == call_name and call.ok), None
|
|
270
|
+
)
|
|
271
|
+
current = found.result if found is not None else None
|
|
272
|
+
for key in parts[2:]:
|
|
273
|
+
if not isinstance(current, dict) or key not in current:
|
|
274
|
+
return value
|
|
275
|
+
current = current[key]
|
|
276
|
+
return current
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _run(
|
|
280
|
+
scenario: Scenario, world_root: Path, *, with_solution: bool
|
|
281
|
+
) -> tuple[GeneratedWorld, list[Call], list[str], list[str]]:
|
|
282
|
+
"""A world set up for this scenario, optionally with the solution played through it."""
|
|
283
|
+
world, _applied, _ready = prepared(scenario, world_root)
|
|
284
|
+
refused: list[str] = []
|
|
285
|
+
assumed: list[str] = []
|
|
286
|
+
if with_solution:
|
|
287
|
+
for step in scenario.solution:
|
|
288
|
+
call = play_reference_step(world, step)
|
|
289
|
+
if not call.ok:
|
|
290
|
+
refused.append(f"{call.name}({step.arguments}): {call.error}")
|
|
291
|
+
runtime_tools = set(getattr(world, "runtime_tools", set()))
|
|
292
|
+
endpoints = getattr(world, "endpoint_for", {}) or {}
|
|
293
|
+
assumed = [
|
|
294
|
+
str(getattr(step, "tool", ""))
|
|
295
|
+
for step in scenario.solution
|
|
296
|
+
if str(getattr(step, "tool", "")) in runtime_tools
|
|
297
|
+
and not endpoints.get(str(getattr(step, "tool", "")))
|
|
298
|
+
]
|
|
299
|
+
if assumed:
|
|
300
|
+
logger.warning(
|
|
301
|
+
"scenario %s: %d of %d solution steps were assumed, not executed (%s). The "
|
|
302
|
+
"proof below covers only the remaining steps.",
|
|
303
|
+
scenario.name,
|
|
304
|
+
len(assumed),
|
|
305
|
+
len(scenario.solution),
|
|
306
|
+
", ".join(sorted(set(assumed))),
|
|
307
|
+
)
|
|
308
|
+
else:
|
|
309
|
+
logger.info(
|
|
310
|
+
"scenario %s: all %d solution steps executed against the world",
|
|
311
|
+
scenario.name,
|
|
312
|
+
len(scenario.solution),
|
|
313
|
+
)
|
|
314
|
+
return world, list(world.calls), refused, sorted(set(assumed))
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def prove(
|
|
318
|
+
scenario: Scenario,
|
|
319
|
+
catalogue: Catalogue,
|
|
320
|
+
world_root: Path,
|
|
321
|
+
*,
|
|
322
|
+
allow_judged_only_with_state: bool = False,
|
|
323
|
+
) -> Proof:
|
|
324
|
+
"""Run all three gates and say whether this scenario is worth keeping."""
|
|
325
|
+
proof = Proof()
|
|
326
|
+
checks = _checks_for(scenario, catalogue)
|
|
327
|
+
if (
|
|
328
|
+
not checks
|
|
329
|
+
and _something_to_check(world_root)
|
|
330
|
+
and not allow_judged_only_with_state
|
|
331
|
+
):
|
|
332
|
+
proof.broken = [
|
|
333
|
+
"none of this sub-goal's checks is in code, and this world has state a check could "
|
|
334
|
+
"read, so settle what happened by reading it rather than by asking a model"
|
|
335
|
+
]
|
|
336
|
+
return proof
|
|
337
|
+
if not checks:
|
|
338
|
+
# Nothing observable to check: no store, no bound endpoint. That is the connect-only shape,
|
|
339
|
+
# an agent hosted elsewhere whose tools we cannot see, and there the conversation is the
|
|
340
|
+
# only evidence there is. Judged sub-goals are the honest grading, so the scenario is
|
|
341
|
+
# allowed through with the gates that still mean something: setup and ready ran, and
|
|
342
|
+
# nothing claims to have been settled by code.
|
|
343
|
+
world, applied, ready = prepared(scenario, world_root)
|
|
344
|
+
world.close()
|
|
345
|
+
proof.why_not_ready = (
|
|
346
|
+
"" if applied.ok and ready.ok else (applied.said or ready.said)
|
|
347
|
+
)
|
|
348
|
+
proof.ready = applied.ok and ready.ok
|
|
349
|
+
proof.solvable = proof.ready
|
|
350
|
+
proof.vacuous = False
|
|
351
|
+
proof.judged_only = True
|
|
352
|
+
return proof
|
|
353
|
+
|
|
354
|
+
# Gate 1: is the world ready for this scenario at all?
|
|
355
|
+
world, applied, ready = prepared(scenario, world_root)
|
|
356
|
+
world.close()
|
|
357
|
+
if not applied.ok:
|
|
358
|
+
proof.why_not_ready = applied.said
|
|
359
|
+
if applied.broken:
|
|
360
|
+
proof.broken = [applied.said]
|
|
361
|
+
return proof
|
|
362
|
+
if not ready.ok:
|
|
363
|
+
proof.why_not_ready = ready.said
|
|
364
|
+
if ready.broken:
|
|
365
|
+
proof.broken = [ready.said]
|
|
366
|
+
return proof
|
|
367
|
+
proof.ready = True
|
|
368
|
+
|
|
369
|
+
# Gate 2: does the reference solution pass this scenario's own checks?
|
|
370
|
+
world, calls, refused, assumed = _run(scenario, world_root, with_solution=True)
|
|
371
|
+
# What the proof below could not cover. Detected here since forever and logged into a void; a
|
|
372
|
+
# scenario whose whole solution was assumed used to save as proved.
|
|
373
|
+
proof.assumed = assumed
|
|
374
|
+
try:
|
|
375
|
+
proof.with_solution = [
|
|
376
|
+
run_check(source, world, calls, name=name) for name, source in checks
|
|
377
|
+
]
|
|
378
|
+
finally:
|
|
379
|
+
world.close()
|
|
380
|
+
proof.refused = refused
|
|
381
|
+
proof.broken = [one.name for one in proof.with_solution if one.broken]
|
|
382
|
+
proof.solvable = all(one.held for one in proof.with_solution) and not proof.broken
|
|
383
|
+
|
|
384
|
+
# Gate 3: do those same checks fail when nothing is done?
|
|
385
|
+
untouched, nothing, _, _ = _run(scenario, world_root, with_solution=False)
|
|
386
|
+
try:
|
|
387
|
+
proof.with_nothing = [
|
|
388
|
+
run_check(source, untouched, nothing, name=name) for name, source in checks
|
|
389
|
+
]
|
|
390
|
+
# A call-presence check can pass the no-op gate yet credit failed operations.
|
|
391
|
+
# Keep the arguments, remove all results/effects, and inject an execution failure.
|
|
392
|
+
# This is not a business refusal: a timeout does not prove correct refusal either.
|
|
393
|
+
crashed = [
|
|
394
|
+
replace(
|
|
395
|
+
call, ok=False, result=None, error="execution failed", refused=False
|
|
396
|
+
)
|
|
397
|
+
for call in calls
|
|
398
|
+
]
|
|
399
|
+
if crashed:
|
|
400
|
+
proof.failed_attempts = [
|
|
401
|
+
name
|
|
402
|
+
for name, source in checks
|
|
403
|
+
if not any(one.name == name and one.held for one in proof.with_nothing)
|
|
404
|
+
and run_check(source, untouched, crashed, name=name).held
|
|
405
|
+
]
|
|
406
|
+
finally:
|
|
407
|
+
untouched.close()
|
|
408
|
+
# Record every check that passes without an action. Even when another checkpoint makes the
|
|
409
|
+
# overall scenario non-vacuous, showing this one as green would award unsupported credit —
|
|
410
|
+
# exactly the misleading partial-pass display the proof gate exists to prevent.
|
|
411
|
+
proof.weak = [one.name for one in proof.with_nothing if one.held]
|
|
412
|
+
# A judged sub-goal reads what the agent said, and an agent that did nothing said nothing, so
|
|
413
|
+
# it cannot be passed by an empty run the way a state check can. It still does not excuse a
|
|
414
|
+
# deterministic checkpoint that independently awards credit with no evidence.
|
|
415
|
+
judged = [
|
|
416
|
+
name
|
|
417
|
+
for name in scenario.sub_goals
|
|
418
|
+
if (found := catalogue.named(name)) is not None and not found.deterministic()
|
|
419
|
+
]
|
|
420
|
+
proof.vacuous = (
|
|
421
|
+
bool(proof.with_nothing)
|
|
422
|
+
and len(proof.weak) == len(proof.with_nothing)
|
|
423
|
+
and not judged
|
|
424
|
+
)
|
|
425
|
+
return proof
|