agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,616 @@
|
|
|
1
|
+
"""The runtime a generated world runs on.
|
|
2
|
+
|
|
3
|
+
A generated world is a database plus one handler per tool. The handler decides what a call does;
|
|
4
|
+
this decides what a handler is allowed to be, what happens when one fails, and what the world
|
|
5
|
+
looks like afterwards. Keeping that here means a generated file stays small enough to read and
|
|
6
|
+
correct, and the parts that must be exact are not regenerated every time.
|
|
7
|
+
|
|
8
|
+
The contract with the rest of the platform is ``EnvironmentAdapter``: ``reset`` publishes the
|
|
9
|
+
tools and the starting state, ``handle_tool_call`` executes one call, and the state afterwards is
|
|
10
|
+
what the checks grade. A world is therefore drivable by any loop that already drives an
|
|
11
|
+
environment, which is the whole reason we generate against this interface rather than inventing
|
|
12
|
+
one.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import copy
|
|
18
|
+
import json
|
|
19
|
+
import re
|
|
20
|
+
import sqlite3
|
|
21
|
+
import time
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any, Mapping, Sequence
|
|
25
|
+
|
|
26
|
+
from ..environment import EnvironmentAdapter, EnvironmentSnapshot, ToolExecutionResult
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ToolError(Exception):
|
|
30
|
+
"""A tool refusing for a real reason the agent should see and recover from.
|
|
31
|
+
|
|
32
|
+
Distinct from a crash. A refusal is the world working: the id does not exist, the item is
|
|
33
|
+
unavailable, the argument is outside what the tool accepts. A crash is our bug, and the two
|
|
34
|
+
must never look the same to a caller deciding whether the agent behaved correctly.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class Db:
|
|
40
|
+
"""The handle a handler gets. Deliberately small: query, execute, one.
|
|
41
|
+
|
|
42
|
+
Handlers get a database, not a filesystem and not a network. Anything a handler can reach is
|
|
43
|
+
something a generated world could depend on, and a world that depends on the outside is not
|
|
44
|
+
reproducible.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
# Whatever this agent's records live in. A handler's statements are written in that store's
|
|
48
|
+
# own language, so this passes them through rather than interpreting them.
|
|
49
|
+
store: Any
|
|
50
|
+
# The agent's own state object, where its tools keep what they act on in memory rather than
|
|
51
|
+
# in a database. Their code is the thing that shapes it, so the world holds it and does not
|
|
52
|
+
# interpret it: freezing it is a serialisation, and restoring it is the reverse.
|
|
53
|
+
state: Any = None
|
|
54
|
+
|
|
55
|
+
def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
|
|
56
|
+
return self.store.query(sql, params)
|
|
57
|
+
|
|
58
|
+
def one(self, sql: str, params: Sequence[Any] = ()) -> dict[str, Any] | None:
|
|
59
|
+
rows = self.query(sql, params)
|
|
60
|
+
return rows[0] if rows else None
|
|
61
|
+
|
|
62
|
+
def execute(self, sql: str, params: Sequence[Any] = ()) -> int:
|
|
63
|
+
return self.store.execute(sql, params)
|
|
64
|
+
|
|
65
|
+
# -- reading without a query language ---------------------------------------------
|
|
66
|
+
#
|
|
67
|
+
# Not every agent has a database. One whose state lives in services and files gets a world
|
|
68
|
+
# whose collections the harness invented, and there is no dialect to write a SELECT in. A
|
|
69
|
+
# handler that could only issue SQL would be unable to read the world it was given at all.
|
|
70
|
+
|
|
71
|
+
def collections(self) -> list[str]:
|
|
72
|
+
"""Every collection this world holds, by name."""
|
|
73
|
+
return list(self.store.collections())
|
|
74
|
+
|
|
75
|
+
def records(self, collection: str) -> list[dict[str, Any]]:
|
|
76
|
+
"""Every record in one collection. The store-agnostic way to read."""
|
|
77
|
+
return list(self.store.records(collection))
|
|
78
|
+
|
|
79
|
+
def find(self, collection: str, **fields: Any) -> list[dict[str, Any]]:
|
|
80
|
+
"""The records in a collection whose fields all match what was asked for."""
|
|
81
|
+
return [
|
|
82
|
+
record
|
|
83
|
+
for record in self.records(collection)
|
|
84
|
+
if all(record.get(field) == value for field, value in fields.items())
|
|
85
|
+
]
|
|
86
|
+
|
|
87
|
+
def add(self, collection: str, record: Mapping[str, Any]) -> int | dict[str, Any]:
|
|
88
|
+
return self.store.add(collection, record)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def settled(value: Any) -> Any:
|
|
92
|
+
"""The value, with a coroutine run to completion first.
|
|
93
|
+
|
|
94
|
+
A tool the agent wrote may well be async: every framework-decorated tool is. Handlers here
|
|
95
|
+
are synchronous, and the build stage is itself inside a running event loop, so ``asyncio.run``
|
|
96
|
+
cannot be called directly. Running it on a worker thread gives it a loop of its own and keeps
|
|
97
|
+
the handler contract unchanged.
|
|
98
|
+
"""
|
|
99
|
+
import asyncio
|
|
100
|
+
import inspect
|
|
101
|
+
|
|
102
|
+
if not inspect.isawaitable(value):
|
|
103
|
+
return value
|
|
104
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
105
|
+
|
|
106
|
+
with ThreadPoolExecutor(max_workers=1) as pool:
|
|
107
|
+
return pool.submit(asyncio.run, value).result()
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _is_refusal(raised: BaseException, refusal_signature: str = "") -> bool:
|
|
111
|
+
"""Whether an exception is the world saying no, rather than the world falling over.
|
|
112
|
+
|
|
113
|
+
Matched by name as well as by identity. A generated handler often declares its own
|
|
114
|
+
``ToolError`` rather than using the one already in scope, which is defensive and sensible
|
|
115
|
+
from where it sits, and would otherwise turn every deliberate refusal into a reported crash.
|
|
116
|
+
Relying on an invisible convention being followed is not a way to decide something this
|
|
117
|
+
load-bearing.
|
|
118
|
+
"""
|
|
119
|
+
if isinstance(raised, ToolError):
|
|
120
|
+
return True
|
|
121
|
+
classes = [base.__name__ for base in type(raised).__mro__]
|
|
122
|
+
if "ToolError" in classes:
|
|
123
|
+
return True
|
|
124
|
+
# Submitted tools commonly use their own semantic exception type (for example
|
|
125
|
+
# ``LookupError`` for a missing account) rather than importing the harness's ToolError.
|
|
126
|
+
# Treat it as a refusal only when source understanding explicitly recorded that exception
|
|
127
|
+
# in the contract. Broadly classifying LookupError/ValueError would hide handler defects;
|
|
128
|
+
# grounding this in the refusal signature preserves the refusal-versus-crash boundary.
|
|
129
|
+
described = refusal_signature or ""
|
|
130
|
+
semantic_classes = [
|
|
131
|
+
name
|
|
132
|
+
for name in classes
|
|
133
|
+
if name not in {"BaseException", "Exception"} and name.endswith("Error")
|
|
134
|
+
]
|
|
135
|
+
return any(
|
|
136
|
+
re.search(rf"\b{re.escape(name)}\b", described) is not None
|
|
137
|
+
for name in semantic_classes
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@dataclass
|
|
142
|
+
class Call:
|
|
143
|
+
"""One tool call and what the world did with it."""
|
|
144
|
+
|
|
145
|
+
name: str
|
|
146
|
+
arguments: dict[str, Any]
|
|
147
|
+
result: Any = None
|
|
148
|
+
ok: bool = True
|
|
149
|
+
error: str = ""
|
|
150
|
+
refused: bool = False
|
|
151
|
+
# When it happened, seconds since the epoch. What lets a recording and a list of calls be
|
|
152
|
+
# read as one thing: without it the UI can show what the agent did but not when, and "when"
|
|
153
|
+
# is the whole question for a spoken run.
|
|
154
|
+
at: float = 0.0
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class GeneratedWorld(EnvironmentAdapter):
|
|
158
|
+
"""A database-backed world whose tools are generated per agent.
|
|
159
|
+
|
|
160
|
+
Subclasses declare ``name``, ``tools`` and ``handlers``. Everything about execution,
|
|
161
|
+
refusal, and state reporting is here so that a generated subclass carries only the parts
|
|
162
|
+
that are specific to one agent.
|
|
163
|
+
"""
|
|
164
|
+
|
|
165
|
+
name = "generated"
|
|
166
|
+
tools: list[dict[str, Any]] = []
|
|
167
|
+
handlers: dict[str, str] = {}
|
|
168
|
+
|
|
169
|
+
# Where the agent's own code lives, so a handler that binds to one of its tools can import
|
|
170
|
+
# it. Empty when the world implements the tools itself.
|
|
171
|
+
source_root: str = ""
|
|
172
|
+
# The agent's own in-memory state, for tools that take it as an argument instead of
|
|
173
|
+
# connecting to anything. Held opaquely: their code gives it shape.
|
|
174
|
+
state_object: Any = None
|
|
175
|
+
# How this agent says no in a returned value. A tool that answers "Error: no such order" is
|
|
176
|
+
# refusing, and recording that as a success would hide the very behaviour worth testing.
|
|
177
|
+
refusal_signature: str = ""
|
|
178
|
+
|
|
179
|
+
def __init__(
|
|
180
|
+
self, database: str | Path = ":memory:", *, store: Any = None, kind: str = ""
|
|
181
|
+
) -> None:
|
|
182
|
+
from .stores import open_store
|
|
183
|
+
|
|
184
|
+
# Generated subclasses declare these as class-level templates. Every world must own
|
|
185
|
+
# its copies: binding and probing mutate them, and sharing the template lets one
|
|
186
|
+
# agent's tools leak into a later agent authored by the same worker process.
|
|
187
|
+
self.tools = copy.deepcopy(type(self).tools)
|
|
188
|
+
self.handlers = dict(type(self).handlers)
|
|
189
|
+
self.state_object = copy.deepcopy(type(self).state_object)
|
|
190
|
+
self.database = str(database)
|
|
191
|
+
# Where this agent's records live. Given rather than assumed, because the harness
|
|
192
|
+
# writes statements in whatever the agent's own store speaks and they have to reach
|
|
193
|
+
# it. A world with no store of its own gets one that says so.
|
|
194
|
+
self.store = store or open_store(kind or "sqlite", database=self.database)
|
|
195
|
+
# A world is usable as soon as it is constructed. SQLite happens to open its
|
|
196
|
+
# connection in __init__, which used to hide this missing lifecycle step; container
|
|
197
|
+
# stores (Postgres, MySQL, …) do not have an address until start() is called.
|
|
198
|
+
# Starting here also makes snapshot.restore() safe, since load_from() immediately
|
|
199
|
+
# writes the frozen rows into the newly-created store.
|
|
200
|
+
self.store.start()
|
|
201
|
+
self.calls: list[Call] = []
|
|
202
|
+
|
|
203
|
+
@property
|
|
204
|
+
def connection(self) -> Any:
|
|
205
|
+
"""The store's own connection, where it has one.
|
|
206
|
+
|
|
207
|
+
Kept so that code written when every world was a SQLite file still works. Anything
|
|
208
|
+
new should go through the store, or through put, change and drop, so it holds for a
|
|
209
|
+
world whose records are somewhere else.
|
|
210
|
+
"""
|
|
211
|
+
found = getattr(self.store, "connection", None)
|
|
212
|
+
if found is None:
|
|
213
|
+
raise AttributeError(
|
|
214
|
+
f"this world's store ({getattr(self.store, 'key', 'unknown')}) has no "
|
|
215
|
+
"connection. Use the store, or put, change and drop."
|
|
216
|
+
)
|
|
217
|
+
return found
|
|
218
|
+
|
|
219
|
+
def reach(self, source_root: str) -> None:
|
|
220
|
+
"""Make the agent's own code importable, so a binding can call it rather than copy it.
|
|
221
|
+
|
|
222
|
+
Two directories go on the path, not one. An agent pointed at flatly is imported from where
|
|
223
|
+
it sits, but an agent laid out as a package is nearly always pointed at the part under
|
|
224
|
+
test rather than at its root: `tau_bench/envs/retail` is where the agent is, while
|
|
225
|
+
`tau_bench.envs.retail.data` only resolves from the repository above it. Adding just the
|
|
226
|
+
directory named makes every import the agent's own code writes fail, which arrives as
|
|
227
|
+
"No module named tau_bench" and reads as the package being absent rather than as us
|
|
228
|
+
having pointed at the middle of it.
|
|
229
|
+
|
|
230
|
+
The package root is found the way Python finds it: walk up while each directory is itself
|
|
231
|
+
a package, and stop at the first that is not.
|
|
232
|
+
"""
|
|
233
|
+
import sys
|
|
234
|
+
|
|
235
|
+
self.source_root = str(source_root or "")
|
|
236
|
+
for path in self._import_roots(self.source_root):
|
|
237
|
+
if path not in sys.path:
|
|
238
|
+
sys.path.insert(0, path)
|
|
239
|
+
|
|
240
|
+
@staticmethod
|
|
241
|
+
def _import_roots(source_root: str) -> list[str]:
|
|
242
|
+
"""Where the agent's code can be imported from: where it sits, and its package root."""
|
|
243
|
+
if not source_root:
|
|
244
|
+
return []
|
|
245
|
+
roots = [source_root]
|
|
246
|
+
here = Path(source_root)
|
|
247
|
+
# Bounded by the filesystem root: `parents` stops there, so a source outside any package
|
|
248
|
+
# simply never enters the loop.
|
|
249
|
+
while (here / "__init__.py").exists() and here.parent != here:
|
|
250
|
+
here = here.parent
|
|
251
|
+
if str(here) not in roots:
|
|
252
|
+
roots.append(str(here))
|
|
253
|
+
return roots
|
|
254
|
+
|
|
255
|
+
# -- EnvironmentAdapter ----------------------------------------------------------
|
|
256
|
+
|
|
257
|
+
def reset(self, **_context: Any) -> EnvironmentSnapshot:
|
|
258
|
+
self.calls = []
|
|
259
|
+
return EnvironmentSnapshot(tools=list(self.tools), state=self.state())
|
|
260
|
+
|
|
261
|
+
def observe(self, **_context: Any) -> EnvironmentSnapshot:
|
|
262
|
+
return EnvironmentSnapshot(tools=list(self.tools), state=self.state())
|
|
263
|
+
|
|
264
|
+
def handle_tool_call(
|
|
265
|
+
self, tool_call: Mapping[str, Any], **_context: Any
|
|
266
|
+
) -> ToolExecutionResult | None:
|
|
267
|
+
name = str(
|
|
268
|
+
tool_call.get("name") or (tool_call.get("function") or {}).get("name") or ""
|
|
269
|
+
)
|
|
270
|
+
call_id = tool_call.get("id") or tool_call.get("tool_call_id")
|
|
271
|
+
arguments = tool_call.get("arguments") or tool_call.get("args") or {}
|
|
272
|
+
if not isinstance(arguments, Mapping):
|
|
273
|
+
arguments = {}
|
|
274
|
+
|
|
275
|
+
call = self.call(name, arguments)
|
|
276
|
+
content = (
|
|
277
|
+
json.dumps(call.result, default=str)
|
|
278
|
+
if not isinstance(call.result, str)
|
|
279
|
+
else call.result
|
|
280
|
+
)
|
|
281
|
+
return ToolExecutionResult(
|
|
282
|
+
tool_call_id=call_id,
|
|
283
|
+
tool_name=name or "unknown",
|
|
284
|
+
content=call.error if not call.ok else content,
|
|
285
|
+
result=call.result,
|
|
286
|
+
success=call.ok,
|
|
287
|
+
error=call.error or None,
|
|
288
|
+
state_updates=self.state(),
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
# -- execution -------------------------------------------------------------------
|
|
292
|
+
|
|
293
|
+
def call(self, name: str, arguments: Mapping[str, Any] | None = None) -> Call:
|
|
294
|
+
"""Execute one call and record it. Never raises: a failure is an outcome, not an event.
|
|
295
|
+
|
|
296
|
+
An unknown tool is a refusal rather than a silent success. An agent reaching for a tool
|
|
297
|
+
that does not exist is a finding, and answering it with an acknowledgement is how a test
|
|
298
|
+
passes something it should have caught.
|
|
299
|
+
"""
|
|
300
|
+
args = dict(arguments or {})
|
|
301
|
+
if name not in self.handlers:
|
|
302
|
+
return self._record(
|
|
303
|
+
Call(
|
|
304
|
+
name=name,
|
|
305
|
+
arguments=args,
|
|
306
|
+
ok=False,
|
|
307
|
+
refused=True,
|
|
308
|
+
error=(
|
|
309
|
+
f"no such tool {name!r}; this agent has "
|
|
310
|
+
f"{', '.join(sorted(self.handlers)) or 'none'}"
|
|
311
|
+
),
|
|
312
|
+
)
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
namespace: dict[str, Any] = {"ToolError": ToolError, "json": json}
|
|
316
|
+
try:
|
|
317
|
+
exec(compile(self.handlers[name], f"<handler:{name}>", "exec"), namespace)
|
|
318
|
+
handle = namespace.get("handle")
|
|
319
|
+
if not callable(handle):
|
|
320
|
+
raise RuntimeError("handler defines no handle(args, db)")
|
|
321
|
+
value = handle(args, Db(self.store, self.state_object))
|
|
322
|
+
except Exception as raised:
|
|
323
|
+
if _is_refusal(raised, self.refusal_signature):
|
|
324
|
+
return self._record(
|
|
325
|
+
Call(
|
|
326
|
+
name=name,
|
|
327
|
+
arguments=args,
|
|
328
|
+
ok=False,
|
|
329
|
+
refused=True,
|
|
330
|
+
error=str(raised),
|
|
331
|
+
)
|
|
332
|
+
)
|
|
333
|
+
# Our bug, not the agent's. Labelled differently so a run is never scored
|
|
334
|
+
# against a world that fell over.
|
|
335
|
+
return self._record(
|
|
336
|
+
Call(
|
|
337
|
+
name=name,
|
|
338
|
+
arguments=args,
|
|
339
|
+
ok=False,
|
|
340
|
+
error=f"{type(raised).__name__}: {raised}",
|
|
341
|
+
)
|
|
342
|
+
)
|
|
343
|
+
# A tool of the agent's own may refuse by returning rather than by raising, which is
|
|
344
|
+
# ordinary in code that was never written to be tested. Recording that as a success
|
|
345
|
+
# would hide exactly the behaviour worth measuring, so the agent's own convention
|
|
346
|
+
# decides. Only the recording differs: the value still reaches the agent unchanged.
|
|
347
|
+
if self._refused_by_value(value):
|
|
348
|
+
return self._record(
|
|
349
|
+
Call(
|
|
350
|
+
name=name,
|
|
351
|
+
arguments=args,
|
|
352
|
+
result=value,
|
|
353
|
+
ok=False,
|
|
354
|
+
refused=True,
|
|
355
|
+
error=str(value)[:400],
|
|
356
|
+
)
|
|
357
|
+
)
|
|
358
|
+
return self._record(Call(name=name, arguments=args, result=value))
|
|
359
|
+
|
|
360
|
+
def _refused_by_value(self, value: Any) -> bool:
|
|
361
|
+
"""Whether a returned value is this agent's way of saying no.
|
|
362
|
+
|
|
363
|
+
The convention is recorded as a description, because that is what somebody reading the
|
|
364
|
+
agent's code can actually write: "strings starting with Error:". So the marker is taken
|
|
365
|
+
from inside it rather than treating the whole sentence as a prefix, which would match
|
|
366
|
+
nothing and quietly record every refusal as a success.
|
|
367
|
+
"""
|
|
368
|
+
if not isinstance(value, str) or not value:
|
|
369
|
+
return False
|
|
370
|
+
described = (self.refusal_signature or "").strip()
|
|
371
|
+
if not described:
|
|
372
|
+
return False
|
|
373
|
+
for marker in self._markers(described):
|
|
374
|
+
if value.lower().startswith(marker.lower()):
|
|
375
|
+
return True
|
|
376
|
+
return False
|
|
377
|
+
|
|
378
|
+
def _markers(self, described: str) -> list[str]:
|
|
379
|
+
"""The literal markers named inside a described convention.
|
|
380
|
+
|
|
381
|
+
Anything quoted is taken as written, since that is how a convention gets spelled out. With
|
|
382
|
+
nothing quoted the whole description is treated as the marker, which is right when somebody
|
|
383
|
+
recorded just the prefix itself.
|
|
384
|
+
"""
|
|
385
|
+
import re
|
|
386
|
+
|
|
387
|
+
# A convention written for people gets quoted the way people quote, and a model writing
|
|
388
|
+
# JSON often escapes those quotes. Left in, the backslash ends up inside the marker, so
|
|
389
|
+
# "Error:" is looked for as 'Error:\' and matches nothing at all. Every refusal is then
|
|
390
|
+
# recorded as a success, which is the failure this whole field exists to prevent.
|
|
391
|
+
plain = described.replace('\\"', '"').replace("\\'", "'")
|
|
392
|
+
quoted = re.findall(r"[\"'“”‘’`]([^\"'“”‘’`]{1,40})[\"'“”‘’`]", plain)
|
|
393
|
+
found = [one.strip().strip("\\").strip() for one in quoted]
|
|
394
|
+
# A convention that lists examples separates them, and the separator sits between one
|
|
395
|
+
# closing quote and the next opening one, so it is matched as though it were quoted too.
|
|
396
|
+
# A marker of "," would make any result beginning with a comma a refusal, so anything
|
|
397
|
+
# without a character a message could start with is dropped.
|
|
398
|
+
found = [one for one in found if any(char.isalnum() for char in one)]
|
|
399
|
+
return found or [plain.strip()]
|
|
400
|
+
|
|
401
|
+
def _record(self, call: Call) -> Call:
|
|
402
|
+
# Stamped here rather than by the caller, so every call is stamped and none of them
|
|
403
|
+
# depend on whoever made it remembering to.
|
|
404
|
+
call.at = call.at or time.time()
|
|
405
|
+
self.calls.append(call)
|
|
406
|
+
return call
|
|
407
|
+
|
|
408
|
+
# -- state -----------------------------------------------------------------------
|
|
409
|
+
|
|
410
|
+
def _settle(self) -> None:
|
|
411
|
+
"""Close any transaction left open on the connection.
|
|
412
|
+
|
|
413
|
+
A handler that only reads still leaves an implicit read transaction behind, and SQLite
|
|
414
|
+
refuses to back up into a connection that has one open: "destination database is in
|
|
415
|
+
use". Left unsettled, the first read-only handler poisons every probe after it, and the
|
|
416
|
+
world can never be checked or saved.
|
|
417
|
+
"""
|
|
418
|
+
connection = getattr(self.store, "connection", None)
|
|
419
|
+
if connection is None:
|
|
420
|
+
# Nothing to settle. A store with no transactions has no open one to close, and
|
|
421
|
+
# reaching for a connection it never had would fail every probe on such a world.
|
|
422
|
+
return
|
|
423
|
+
try:
|
|
424
|
+
connection.commit()
|
|
425
|
+
except sqlite3.Error:
|
|
426
|
+
connection.rollback()
|
|
427
|
+
|
|
428
|
+
def checkpoint(self) -> Any:
|
|
429
|
+
"""A copy of everything the world holds, to come back to.
|
|
430
|
+
|
|
431
|
+
Probes and smoke calls mutate: ordering an item inserts a record, cancelling one changes
|
|
432
|
+
it. Without a way back, each runs against the debris of the ones before it, and a check
|
|
433
|
+
expecting three records finds seven.
|
|
434
|
+
|
|
435
|
+
Both halves are copied, and that matters more for an adopted world than a generated one.
|
|
436
|
+
A tool the agent wrote changes the structure it was given, in place. Backing up only the
|
|
437
|
+
store would leave those changes permanent, so a smoke call against one record would quietly
|
|
438
|
+
spend it, and whatever ran later against that same record would fail for a reason nothing
|
|
439
|
+
could see.
|
|
440
|
+
"""
|
|
441
|
+
import copy as duplicate
|
|
442
|
+
|
|
443
|
+
self._settle()
|
|
444
|
+
# Through the store's own freeze rather than a SQLite backup, so a world whose records
|
|
445
|
+
# live somewhere else is revertible too. Every store knows how to go back; only some of
|
|
446
|
+
# them have a connection to copy.
|
|
447
|
+
store = self.store.freeze()
|
|
448
|
+
held = (
|
|
449
|
+
duplicate.deepcopy(self.state_object)
|
|
450
|
+
if self.state_object is not None
|
|
451
|
+
else None
|
|
452
|
+
)
|
|
453
|
+
return {"store": store, "state": held}
|
|
454
|
+
|
|
455
|
+
def revert(self, checkpoint: Any) -> None:
|
|
456
|
+
"""Put everything back as it was when the checkpoint was taken."""
|
|
457
|
+
import copy as duplicate
|
|
458
|
+
|
|
459
|
+
self._settle()
|
|
460
|
+
# A bare connection is accepted so that anything written against the older shape of this
|
|
461
|
+
# method keeps working rather than reverting nothing at all, which would be silent.
|
|
462
|
+
if isinstance(checkpoint, sqlite3.Connection):
|
|
463
|
+
checkpoint.backup(self.connection)
|
|
464
|
+
return
|
|
465
|
+
held = (checkpoint or {}).get("store")
|
|
466
|
+
if held is not None:
|
|
467
|
+
self.store.restore(held)
|
|
468
|
+
if (checkpoint or {}).get("state") is not None:
|
|
469
|
+
self.state_object = duplicate.deepcopy(checkpoint["state"])
|
|
470
|
+
|
|
471
|
+
def state(self) -> dict[str, Any]:
|
|
472
|
+
"""What the checks compare against after a run.
|
|
473
|
+
|
|
474
|
+
Tables and their rows, plus whatever the agent's own tools keep in memory. A world that
|
|
475
|
+
adopted the agent's code may have all of its state in the second of those, so a check has
|
|
476
|
+
to be able to see both without knowing which kind of world it is grading.
|
|
477
|
+
"""
|
|
478
|
+
found: dict[str, Any] = {
|
|
479
|
+
name: self.store.records(name) for name in self.store.collections()
|
|
480
|
+
}
|
|
481
|
+
if isinstance(self.state_object, dict):
|
|
482
|
+
# Collections the agent's own code owns. Not merged blindly: a table and a key of
|
|
483
|
+
# the same name would silently shadow one another, and a check comparing the wrong
|
|
484
|
+
# one would be wrong in a way nobody could see.
|
|
485
|
+
for key, value in self.state_object.items():
|
|
486
|
+
found.setdefault(str(key), value)
|
|
487
|
+
elif self.state_object is not None:
|
|
488
|
+
found.setdefault("state", self.state_object)
|
|
489
|
+
return found
|
|
490
|
+
|
|
491
|
+
# -- changing the world, without naming what it is kept in ------------------------
|
|
492
|
+
#
|
|
493
|
+
# A scenario changes the world before it runs, and it must not have to know whether the world
|
|
494
|
+
# is a database, a mapping the agent's own code owns, or something else again. Speaking SQL
|
|
495
|
+
# here would write SQLite into every scenario ever written, and the store is the one thing
|
|
496
|
+
# this design expects to vary per agent.
|
|
497
|
+
#
|
|
498
|
+
# So the vocabulary is collections and records, which every store has under some name, and
|
|
499
|
+
# each method dispatches on what the collection actually is. The preferred way to change the
|
|
500
|
+
# world is still the agent's own tools, because anything they refuse would have refused the
|
|
501
|
+
# agent too; these are for the states no tool can produce.
|
|
502
|
+
|
|
503
|
+
def _table(self, collection: str) -> bool:
|
|
504
|
+
return bool(self.store.holds(collection))
|
|
505
|
+
|
|
506
|
+
def _held(self, collection: str) -> Any:
|
|
507
|
+
if isinstance(self.state_object, dict):
|
|
508
|
+
return self.state_object.get(collection)
|
|
509
|
+
return None
|
|
510
|
+
|
|
511
|
+
def put(self, collection: str, record: Mapping[str, Any], *, key: str = "") -> None:
|
|
512
|
+
"""Add one record to a collection, whatever the collection is kept in."""
|
|
513
|
+
if self._table(collection):
|
|
514
|
+
self.store.add(collection, record)
|
|
515
|
+
return
|
|
516
|
+
held = self._held(collection)
|
|
517
|
+
if isinstance(held, dict):
|
|
518
|
+
if not key:
|
|
519
|
+
raise KeyError(
|
|
520
|
+
f"{collection} is keyed, so adding to it needs a key: "
|
|
521
|
+
"world.put(collection, record, key=...)"
|
|
522
|
+
)
|
|
523
|
+
held[key] = dict(record)
|
|
524
|
+
return
|
|
525
|
+
if isinstance(held, list):
|
|
526
|
+
held.append(dict(record))
|
|
527
|
+
return
|
|
528
|
+
# A collection nobody has created yet is made here rather than refused. An agent whose
|
|
529
|
+
# state lives in services and files has no store to declare tables in, so every collection
|
|
530
|
+
# the world needs is one the harness invents: refusing the first record leaves that agent
|
|
531
|
+
# with a world that cannot hold anything at all.
|
|
532
|
+
made = getattr(self.store, "start_collection", None)
|
|
533
|
+
if callable(made):
|
|
534
|
+
made(collection, keyed=bool(key))
|
|
535
|
+
self.store.add(collection, {**record, "_id": key} if key else record)
|
|
536
|
+
return
|
|
537
|
+
raise KeyError(
|
|
538
|
+
f"no collection called {collection!r}; this world has {sorted(self.state())}"
|
|
539
|
+
)
|
|
540
|
+
|
|
541
|
+
def change(
|
|
542
|
+
self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
|
|
543
|
+
) -> int:
|
|
544
|
+
"""Change records in a collection. Returns how many were changed.
|
|
545
|
+
|
|
546
|
+
``by`` names the column a table is keyed on. A collection the agent's own code keeps is
|
|
547
|
+
keyed already, so it is not needed there.
|
|
548
|
+
"""
|
|
549
|
+
if self._table(collection):
|
|
550
|
+
return self.store.amend(collection, key, changes, by=by)
|
|
551
|
+
held = self._held(collection)
|
|
552
|
+
if isinstance(held, dict) and key in held:
|
|
553
|
+
if isinstance(held[key], dict):
|
|
554
|
+
held[key].update(dict(changes))
|
|
555
|
+
else:
|
|
556
|
+
held[key] = dict(changes)
|
|
557
|
+
return 1
|
|
558
|
+
raise KeyError(f"nothing called {key!r} in {collection!r}")
|
|
559
|
+
|
|
560
|
+
def drop(self, collection: str, key: str = "", *, by: str = "") -> int:
|
|
561
|
+
"""Remove a record, or the whole contents of a collection when no key is given."""
|
|
562
|
+
if self._table(collection):
|
|
563
|
+
return self.store.remove(collection, key, by=by)
|
|
564
|
+
held = self._held(collection)
|
|
565
|
+
if isinstance(held, dict):
|
|
566
|
+
if not key:
|
|
567
|
+
count = len(held)
|
|
568
|
+
held.clear()
|
|
569
|
+
return count
|
|
570
|
+
return 1 if held.pop(key, None) is not None else 0
|
|
571
|
+
if isinstance(held, list):
|
|
572
|
+
count = len(held)
|
|
573
|
+
del held[:]
|
|
574
|
+
return count
|
|
575
|
+
raise KeyError(
|
|
576
|
+
f"no collection called {collection!r}; this world has {sorted(self.state())}"
|
|
577
|
+
)
|
|
578
|
+
|
|
579
|
+
def shapes(self) -> str:
|
|
580
|
+
"""What this world's collections actually are, in words.
|
|
581
|
+
|
|
582
|
+
Said wherever code written against the wrong shape fails. A table gives a list of records;
|
|
583
|
+
a collection the agent's own code keeps is often a mapping keyed by identifier, and
|
|
584
|
+
iterating that yields strings. No amount of general advice substitutes for naming which is
|
|
585
|
+
which, for the world in front of whoever got it wrong.
|
|
586
|
+
"""
|
|
587
|
+
lines = []
|
|
588
|
+
for name, held in sorted(self.state().items()):
|
|
589
|
+
if isinstance(held, dict):
|
|
590
|
+
first = next(iter(held), None)
|
|
591
|
+
lines.append(
|
|
592
|
+
f" {name}: a mapping of {len(held)} records keyed by identifier"
|
|
593
|
+
+ (f", e.g. {first!r}" if first is not None else "")
|
|
594
|
+
+ ". Iterate .values(), or .items() when the key matters."
|
|
595
|
+
)
|
|
596
|
+
elif isinstance(held, list):
|
|
597
|
+
lines.append(
|
|
598
|
+
f" {name}: a list of {len(held)} records. Iterate it directly."
|
|
599
|
+
)
|
|
600
|
+
else:
|
|
601
|
+
lines.append(f" {name}: a single {type(held).__name__}.")
|
|
602
|
+
return "This world holds:\n" + ("\n".join(lines) or " nothing yet")
|
|
603
|
+
|
|
604
|
+
def close(self) -> None:
|
|
605
|
+
self.store.close()
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
@dataclass
|
|
609
|
+
class WorldSpec:
|
|
610
|
+
"""What a generated world is, before it is written out."""
|
|
611
|
+
|
|
612
|
+
agent: str
|
|
613
|
+
schema_sql: str = ""
|
|
614
|
+
tools: list[dict[str, Any]] = field(default_factory=list)
|
|
615
|
+
handlers: dict[str, str] = field(default_factory=dict)
|
|
616
|
+
notes: str = ""
|