agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""Freezing a world, and starting every scenario from the same frozen copy.
|
|
2
|
+
|
|
3
|
+
The database is built once and snapshotted; that snapshot is the base state. A scenario restores
|
|
4
|
+
its own copy and layers on whatever it additionally needs, so scenarios cannot inherit each
|
|
5
|
+
other's leftovers and a run is repeatable a week later.
|
|
6
|
+
|
|
7
|
+
Which is why the overlay exists: a scenario that needs a customer with three open orders adds
|
|
8
|
+
those rows to a restored copy rather than editing the snapshot. The base world stays the shared
|
|
9
|
+
starting point instead of drifting toward whichever scenario was written last.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import shutil
|
|
16
|
+
import sqlite3
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Mapping
|
|
19
|
+
|
|
20
|
+
from .runtime import GeneratedWorld
|
|
21
|
+
|
|
22
|
+
DATABASE = "world.sqlite"
|
|
23
|
+
HANDLERS = "handlers"
|
|
24
|
+
MANIFEST = "manifest.json"
|
|
25
|
+
STATE = "state.json"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def saved(path: str | Path | None) -> bool:
|
|
29
|
+
"""Whether a world has been written here.
|
|
30
|
+
|
|
31
|
+
One function, because this question gets asked from six places: the build stage, the
|
|
32
|
+
conversation, the session listing, the CLI and the UI. Asked as "is there a world.sqlite"
|
|
33
|
+
each of those was really asking "is this a SQLite world", so an agent whose state lives in
|
|
34
|
+
services and files saved a world that scored 1.00 and was then invisible to all of them.
|
|
35
|
+
"""
|
|
36
|
+
return bool(path) and (Path(path) / MANIFEST).exists()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
WORLD_MODULE = "world.py"
|
|
40
|
+
|
|
41
|
+
_MODULE = '''"""Generated world for {agent}. Do not edit by hand; regenerate instead.
|
|
42
|
+
|
|
43
|
+
{notes}
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
|
|
48
|
+
from fi.alk.harness.world.runtime import GeneratedWorld
|
|
49
|
+
|
|
50
|
+
_HERE = Path(__file__).parent
|
|
51
|
+
|
|
52
|
+
TOOLS = {tools}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class World(GeneratedWorld):
|
|
56
|
+
name = {agent!r}
|
|
57
|
+
tools = TOOLS
|
|
58
|
+
handlers = {{
|
|
59
|
+
name: (_HERE / "handlers" / f"{{name}}.py").read_text(encoding="utf-8")
|
|
60
|
+
for name in {handler_names}
|
|
61
|
+
}}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def load(database=None):
|
|
65
|
+
"""This world, restored from the snapshot beside this file.
|
|
66
|
+
|
|
67
|
+
Through `restore` rather than by opening a database directly, because not every world has
|
|
68
|
+
one: an agent whose state lives in services and files keeps its records in the snapshot, and
|
|
69
|
+
naming a SQLite file would hand back an empty world instead of this one.
|
|
70
|
+
"""
|
|
71
|
+
from fi.alk.harness.world.snapshot import restore
|
|
72
|
+
|
|
73
|
+
return restore(_HERE, into=database) if database else restore(_HERE)
|
|
74
|
+
'''
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def save(
|
|
78
|
+
world: GeneratedWorld,
|
|
79
|
+
path: str | Path,
|
|
80
|
+
*,
|
|
81
|
+
notes: str = "",
|
|
82
|
+
sequences: list[dict[str, Any]] | None = None,
|
|
83
|
+
world_checks: Mapping[str, str] | None = None,
|
|
84
|
+
) -> Path:
|
|
85
|
+
"""Write the world out: the snapshot, the handlers, the module, and a manifest."""
|
|
86
|
+
root = Path(path)
|
|
87
|
+
(root / HANDLERS).mkdir(parents=True, exist_ok=True)
|
|
88
|
+
|
|
89
|
+
# Through the store, so a world whose records live somewhere other than a SQLite file, or
|
|
90
|
+
# nowhere at all, freezes by its own means rather than by one assumed here.
|
|
91
|
+
world.store.save_to(root)
|
|
92
|
+
|
|
93
|
+
for name, source in world.handlers.items():
|
|
94
|
+
(root / HANDLERS / f"{name}.py").write_text(source, encoding="utf-8")
|
|
95
|
+
|
|
96
|
+
(root / WORLD_MODULE).write_text(
|
|
97
|
+
_MODULE.format(
|
|
98
|
+
agent=world.name,
|
|
99
|
+
notes=notes or "Generated from the agent's contract.",
|
|
100
|
+
tools=json.dumps(world.tools, indent=4),
|
|
101
|
+
handler_names=json.dumps(sorted(world.handlers)),
|
|
102
|
+
),
|
|
103
|
+
encoding="utf-8",
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# The agent's own in-memory state, where its tools keep what they act on there rather
|
|
107
|
+
# than in the database. Frozen as JSON so restoring is the exact reverse, and so a
|
|
108
|
+
# person can read what the world starts from.
|
|
109
|
+
if world.state_object is not None:
|
|
110
|
+
# Round-tripped rather than only written. Every scenario restores from this file, so state
|
|
111
|
+
# that does not survive the trip would come back subtly different and every check after
|
|
112
|
+
# it would be grading something else. Better to fail here than to be wrong quietly.
|
|
113
|
+
frozen = json.dumps(world.state_object, indent=2, default=str)
|
|
114
|
+
if json.loads(frozen) != world.state_object:
|
|
115
|
+
raise ValueError(
|
|
116
|
+
"the agent's state does not survive being frozen as JSON, so restoring it would "
|
|
117
|
+
"not give back what was saved. Every scenario starts from that restore, so this "
|
|
118
|
+
"world cannot be trusted. What is in the state that is not plain JSON?"
|
|
119
|
+
)
|
|
120
|
+
(root / STATE).write_text(frozen, encoding="utf-8")
|
|
121
|
+
|
|
122
|
+
state = world.state()
|
|
123
|
+
(root / MANIFEST).write_text(
|
|
124
|
+
json.dumps(
|
|
125
|
+
{
|
|
126
|
+
"agent": world.name,
|
|
127
|
+
# Which store this world used, so restoring it opens the same one rather
|
|
128
|
+
# than assuming a database that may never have existed.
|
|
129
|
+
"store": getattr(world.store, "key", "sqlite"),
|
|
130
|
+
"tools": sorted(world.handlers),
|
|
131
|
+
# Written because restore reads it. Without it a restored world publishes no
|
|
132
|
+
# tool descriptions at all, and every later stage has to reconstruct them.
|
|
133
|
+
"tool_specs": list(world.tools),
|
|
134
|
+
"tables": {name: len(rows) for name, rows in state.items()},
|
|
135
|
+
# Kept because they are judgement about this agent, not something a schema
|
|
136
|
+
# implies. A world picked up again can be re-verified without redeclaring them.
|
|
137
|
+
"sequences": list(sequences or []),
|
|
138
|
+
# The world's own checks are judgement about this agent, so a world picked
|
|
139
|
+
# up again keeps them rather than having them rewritten from scratch.
|
|
140
|
+
"world_checks": dict(world_checks or {}),
|
|
141
|
+
# Where the agent's own code lives. Kept because a restored world has to
|
|
142
|
+
# be able to import the tools it was bound to, and a scenario run happens
|
|
143
|
+
# long after the build stage that found the path.
|
|
144
|
+
"source_root": world.source_root,
|
|
145
|
+
# A run refuses legacy/demo worlds whose handlers were authored by the harness.
|
|
146
|
+
# New worlds can only acquire handlers through adopt_tool, and a source root is
|
|
147
|
+
# required for that import to work.
|
|
148
|
+
"tool_implementation": "source" if world.source_root else "synthetic",
|
|
149
|
+
"runtime_tools": sorted(getattr(world, "runtime_tools", set())),
|
|
150
|
+
"external_runtime": bool(getattr(world, "external_runtime", False)),
|
|
151
|
+
# How this agent says no in a returned value. Without it a restored world
|
|
152
|
+
# cannot tell a refusal from a success, so every run records "Error: no such
|
|
153
|
+
# order" as if the call worked, and a check asking whether the agent was
|
|
154
|
+
# refused is answered wrongly rather than reported as unanswerable.
|
|
155
|
+
"refusal_signature": world.refusal_signature,
|
|
156
|
+
"notes": notes,
|
|
157
|
+
},
|
|
158
|
+
indent=2,
|
|
159
|
+
ensure_ascii=False,
|
|
160
|
+
),
|
|
161
|
+
encoding="utf-8",
|
|
162
|
+
)
|
|
163
|
+
return root
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def restore(path: str | Path, *, into: str | Path | None = None) -> GeneratedWorld:
|
|
167
|
+
"""A fresh, independent copy of the frozen world.
|
|
168
|
+
|
|
169
|
+
In memory by default, because a scenario should not be able to write back into the snapshot
|
|
170
|
+
every later scenario depends on.
|
|
171
|
+
"""
|
|
172
|
+
root = Path(path)
|
|
173
|
+
source = root / DATABASE
|
|
174
|
+
if not (root / MANIFEST).exists():
|
|
175
|
+
raise FileNotFoundError(f"no world snapshot at {root}")
|
|
176
|
+
|
|
177
|
+
manifest = read_manifest(root)
|
|
178
|
+
provisioned_http = False
|
|
179
|
+
if (root / "environment.json").exists():
|
|
180
|
+
from ..provision import ProvisionedEnvironment
|
|
181
|
+
|
|
182
|
+
environment = ProvisionedEnvironment.load(root)
|
|
183
|
+
provisioned_http = bool(
|
|
184
|
+
environment
|
|
185
|
+
and any(
|
|
186
|
+
value.startswith(("http://", "https://"))
|
|
187
|
+
for value in environment.overrides.values()
|
|
188
|
+
)
|
|
189
|
+
)
|
|
190
|
+
if provisioned_http:
|
|
191
|
+
if into is not None:
|
|
192
|
+
raise ValueError(
|
|
193
|
+
"a source-provisioned world restores into its submitted service, not a database "
|
|
194
|
+
"file path"
|
|
195
|
+
)
|
|
196
|
+
from ..understand import load as load_contract
|
|
197
|
+
from .provisioned import open_provisioned_world
|
|
198
|
+
|
|
199
|
+
contract = load_contract(root)
|
|
200
|
+
if contract is None:
|
|
201
|
+
raise FileNotFoundError(f"no contract beside source environment at {root}")
|
|
202
|
+
world = open_provisioned_world(
|
|
203
|
+
root,
|
|
204
|
+
contract,
|
|
205
|
+
source_root=str(manifest.get("source_root") or ""),
|
|
206
|
+
)
|
|
207
|
+
world.store.load_from(root)
|
|
208
|
+
world.tools = manifest.get("tool_specs", [])
|
|
209
|
+
world.refusal_signature = str(manifest.get("refusal_signature") or "")
|
|
210
|
+
return world
|
|
211
|
+
handlers = {
|
|
212
|
+
name: (root / HANDLERS / f"{name}.py").read_text(encoding="utf-8")
|
|
213
|
+
for name in manifest.get("tools", [])
|
|
214
|
+
if (root / HANDLERS / f"{name}.py").exists()
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
named = str(manifest.get("store") or "sqlite")
|
|
218
|
+
if into is None:
|
|
219
|
+
world = GeneratedWorld(":memory:", kind=named)
|
|
220
|
+
# Only where there is one. A world whose records the agent's own code keeps has no
|
|
221
|
+
# database file, and demanding one would make it unrestorable.
|
|
222
|
+
if source.exists() and getattr(world.store, "connection", None) is not None:
|
|
223
|
+
origin = sqlite3.connect(source)
|
|
224
|
+
with world.connection:
|
|
225
|
+
origin.backup(world.connection)
|
|
226
|
+
origin.close()
|
|
227
|
+
else:
|
|
228
|
+
# A store that keeps its records somewhere other than a SQLite file loads them its own
|
|
229
|
+
# way. Without this the world comes back with an empty store, and everything a check
|
|
230
|
+
# reads is whatever happened to land in the agent's state instead.
|
|
231
|
+
world.store.load_from(root)
|
|
232
|
+
else:
|
|
233
|
+
target = Path(into)
|
|
234
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
235
|
+
shutil.copyfile(source, target)
|
|
236
|
+
world = GeneratedWorld(target, kind=named)
|
|
237
|
+
|
|
238
|
+
world.name = manifest.get("agent", "generated")
|
|
239
|
+
world.handlers = handlers
|
|
240
|
+
world.runtime_tools = set(manifest.get("runtime_tools") or [])
|
|
241
|
+
world.external_runtime = bool(manifest.get("external_runtime", False))
|
|
242
|
+
world.tools = manifest.get("tool_specs", [])
|
|
243
|
+
world.refusal_signature = str(manifest.get("refusal_signature") or "")
|
|
244
|
+
# A world whose handlers bind to the agent's own code cannot run them unless that code
|
|
245
|
+
# is importable again, and the frozen state is what those tools act on.
|
|
246
|
+
reached = str(manifest.get("source_root") or "")
|
|
247
|
+
if reached:
|
|
248
|
+
world.reach(reached)
|
|
249
|
+
frozen_state = root / STATE
|
|
250
|
+
if frozen_state.exists():
|
|
251
|
+
world.state_object = json.loads(frozen_state.read_text(encoding="utf-8"))
|
|
252
|
+
return world
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def read_manifest(path: str | Path) -> dict[str, Any]:
|
|
256
|
+
return json.loads((Path(path) / MANIFEST).read_text(encoding="utf-8"))
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def require_source_implementation(path: str | Path) -> None:
|
|
260
|
+
"""Refuse worlds that do not prove their tools came from the submitted source."""
|
|
261
|
+
manifest = read_manifest(path)
|
|
262
|
+
if manifest.get("tool_implementation") != "source":
|
|
263
|
+
raise RuntimeError(
|
|
264
|
+
"This environment has no source-implementation provenance. It was created by an "
|
|
265
|
+
"older/synthetic harness path and may contain reimplemented handlers. Rebuild it "
|
|
266
|
+
"from the agent repository; it cannot be used for a test run."
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def apply_overlay(world: GeneratedWorld, overlay: Mapping[str, Any] | None) -> int:
|
|
271
|
+
"""Layer one scenario's own rows onto a restored world.
|
|
272
|
+
|
|
273
|
+
``{"table": [{"column": value}, ...]}``. The only sanctioned way a scenario adds data, so the
|
|
274
|
+
base world stays the shared starting point rather than drifting per scenario.
|
|
275
|
+
"""
|
|
276
|
+
written = 0
|
|
277
|
+
for table, rows in (overlay or {}).items():
|
|
278
|
+
for row in rows or []:
|
|
279
|
+
if not isinstance(row, Mapping) or not row:
|
|
280
|
+
continue
|
|
281
|
+
columns = ", ".join(row)
|
|
282
|
+
marks = ", ".join("?" for _ in row)
|
|
283
|
+
world.connection.execute(
|
|
284
|
+
f"INSERT INTO {table} ({columns}) VALUES ({marks})", list(row.values())
|
|
285
|
+
)
|
|
286
|
+
written += 1
|
|
287
|
+
world.connection.commit()
|
|
288
|
+
return written
|
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""The stores the harness can stand up for an agent, and what every one of them owes a world.
|
|
2
|
+
|
|
3
|
+
A store is the thing underneath an agent's tools: whatever really holds the records its queries
|
|
4
|
+
run against. It is never asked to execute a tool. It is asked to exist, to hold data, to say what
|
|
5
|
+
it holds, to let a scenario change a little of it, and to go back to how it was.
|
|
6
|
+
|
|
7
|
+
Which engine gets stood up is read off the agent, never chosen for it. Postgres and ClickHouse
|
|
8
|
+
disagree about dialect, types and what a transaction even means, so testing one against the other
|
|
9
|
+
grades an agent on queries it never runs. An engine the harness cannot stand up is an answer, not
|
|
10
|
+
a reason to substitute something that merely resembles it.
|
|
11
|
+
|
|
12
|
+
What a store owes falls into four groups, and most stores care about three:
|
|
13
|
+
|
|
14
|
+
lifecycle start, stop, dsn stand it up and say where it is
|
|
15
|
+
contents apply, execute, query statements, in whatever this engine speaks
|
|
16
|
+
records collections, holds, records, add, amend, remove
|
|
17
|
+
going back freeze, restore between scenarios
|
|
18
|
+
save_to, load_from to and from disk, for the base world
|
|
19
|
+
|
|
20
|
+
The records group is what keeps a scenario from ever naming a store. `world.put`, `world.change`
|
|
21
|
+
and `world.drop` land here, so the same scenario runs against SQLite, against Postgres in a
|
|
22
|
+
container, or against a structure the agent's own code holds, without a line of it changing.
|
|
23
|
+
`state()` comes free from that group, and `Records` provides it.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
from typing import Any, Callable, Mapping, Protocol, Sequence, runtime_checkable
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class StoreError(RuntimeError):
|
|
34
|
+
"""The store could not be stood up, or could not answer.
|
|
35
|
+
|
|
36
|
+
Distinct from anything the agent did. A store that will not start is our problem and should
|
|
37
|
+
stop the run loudly, because every result after it would be measured against something that
|
|
38
|
+
is not there.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Snapshot:
|
|
44
|
+
"""Everything a store held at one moment, and what it takes to put it back.
|
|
45
|
+
|
|
46
|
+
``rows`` is kept in the shape ``state()`` reports, so a check written against a world's state
|
|
47
|
+
reads a snapshot without knowing which engine produced it.
|
|
48
|
+
|
|
49
|
+
``counters`` is whatever an engine hands out that is not itself a record: a Postgres sequence,
|
|
50
|
+
a MySQL auto-increment, anything that keeps counting after the rows are gone. Restoring rows
|
|
51
|
+
without restoring these gives the next scenario ids that continue from the last one, and a
|
|
52
|
+
check naming a specific id then fails for a reason that has nothing to do with the agent.
|
|
53
|
+
Engines that hand out nothing of the sort leave it empty, which is not a gap.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
rows: dict[str, list[dict[str, Any]]] = field(default_factory=dict)
|
|
57
|
+
counters: dict[str, int] = field(default_factory=dict)
|
|
58
|
+
|
|
59
|
+
def counts(self) -> dict[str, int]:
|
|
60
|
+
return {name: len(rows) for name, rows in self.rows.items()}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@runtime_checkable
|
|
64
|
+
class Store(Protocol):
|
|
65
|
+
"""A running store the world's records live in."""
|
|
66
|
+
|
|
67
|
+
# What this engine is. ``key`` is the same thing under the name a saved manifest already
|
|
68
|
+
# uses, so a world written before this split still reopens.
|
|
69
|
+
engine: str
|
|
70
|
+
key: str
|
|
71
|
+
|
|
72
|
+
def start(self) -> None: ...
|
|
73
|
+
def stop(self) -> None: ...
|
|
74
|
+
def dsn(self) -> str: ...
|
|
75
|
+
|
|
76
|
+
# Statements the harness wrote, in whatever this store speaks.
|
|
77
|
+
def apply(self, script: str) -> None: ...
|
|
78
|
+
def execute(self, statement: str, params: Sequence[Any] = ()) -> int: ...
|
|
79
|
+
def query(
|
|
80
|
+
self, statement: str, params: Sequence[Any] = ()
|
|
81
|
+
) -> list[dict[str, Any]]: ...
|
|
82
|
+
|
|
83
|
+
# What a scenario and its checks need without writing a statement themselves.
|
|
84
|
+
def collections(self) -> list[str]: ...
|
|
85
|
+
def holds(self, collection: str) -> bool: ...
|
|
86
|
+
def records(self, collection: str) -> list[dict[str, Any]]: ...
|
|
87
|
+
def state(self) -> dict[str, list[dict[str, Any]]]: ...
|
|
88
|
+
def table(self, name: str) -> list[dict[str, Any]]: ...
|
|
89
|
+
def add(self, collection: str, record: Mapping[str, Any]) -> int | dict[str, Any]: ...
|
|
90
|
+
def amend(
|
|
91
|
+
self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
|
|
92
|
+
) -> int: ...
|
|
93
|
+
def remove(self, collection: str, key: str = "", *, by: str = "") -> int: ...
|
|
94
|
+
|
|
95
|
+
# Between scenarios, in memory.
|
|
96
|
+
def freeze(self) -> Snapshot: ...
|
|
97
|
+
def restore(self, snapshot: Snapshot) -> None: ...
|
|
98
|
+
|
|
99
|
+
# To and from disk, so the base world outlives the process that built it.
|
|
100
|
+
def save_to(self, path: str | Path) -> None: ...
|
|
101
|
+
def load_from(self, path: str | Path) -> None: ...
|
|
102
|
+
|
|
103
|
+
def close(self) -> None: ...
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class Records:
|
|
107
|
+
"""``state`` from the record methods, for any store that has them.
|
|
108
|
+
|
|
109
|
+
Kept in one place because the two would otherwise drift, and they are the pair the gates
|
|
110
|
+
compare: the bite gate empties a store and reads ``state``, while a scenario changes it
|
|
111
|
+
through ``add`` and ``amend``. If those disagree about what a collection contains, a check
|
|
112
|
+
passes against something no scenario can produce.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
def state(self) -> dict[str, list[dict[str, Any]]]:
|
|
116
|
+
return {name: self.records(name) for name in self.collections()} # type: ignore[attr-defined]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class Held:
|
|
120
|
+
"""The record methods, and disk, for a store that already answers ``state``.
|
|
121
|
+
|
|
122
|
+
The mirror of ``Records``, for stores built the other way round: a container store reads
|
|
123
|
+
everything it holds in one go, and the per-collection questions follow from that. Saving to
|
|
124
|
+
disk is the snapshot as JSON, which works for any engine because a snapshot is already the
|
|
125
|
+
engine-independent shape.
|
|
126
|
+
|
|
127
|
+
``add``, ``amend`` and ``remove`` are not derivable and are left to the engine. A store
|
|
128
|
+
without them refuses loudly rather than silently doing nothing, because the alternative is a
|
|
129
|
+
scenario whose setup appears to run and changes nothing, and a run then graded against a
|
|
130
|
+
world that was never set up.
|
|
131
|
+
"""
|
|
132
|
+
|
|
133
|
+
engine: str = ""
|
|
134
|
+
|
|
135
|
+
@property
|
|
136
|
+
def key(self) -> str:
|
|
137
|
+
return self.engine
|
|
138
|
+
|
|
139
|
+
def collections(self) -> list[str]:
|
|
140
|
+
return sorted(self.state()) # type: ignore[attr-defined]
|
|
141
|
+
|
|
142
|
+
def holds(self, collection: str) -> bool:
|
|
143
|
+
return collection in self.state() # type: ignore[attr-defined]
|
|
144
|
+
|
|
145
|
+
def records(self, collection: str) -> list[dict[str, Any]]:
|
|
146
|
+
return self.state().get(collection, []) # type: ignore[attr-defined]
|
|
147
|
+
|
|
148
|
+
def execute(self, statement: str, params: Sequence[Any] = ()) -> int:
|
|
149
|
+
self.apply(statement) # type: ignore[attr-defined]
|
|
150
|
+
return 0
|
|
151
|
+
|
|
152
|
+
def query(self, statement: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
|
|
153
|
+
raise StoreError(
|
|
154
|
+
f"{self.engine} does not read back arbitrary statements. Read what it holds with "
|
|
155
|
+
"records() or state()."
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
def table(self, name: str) -> list[dict[str, Any]]:
|
|
159
|
+
raise StoreError(
|
|
160
|
+
f"{self.engine} does not read one table at a time. Read what it holds with "
|
|
161
|
+
"records() or state()."
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
def add(self, collection: str, record: Mapping[str, Any]) -> int | dict[str, Any]:
|
|
165
|
+
raise StoreError(_UNWRITABLE.format(engine=self.engine, verb="add to"))
|
|
166
|
+
|
|
167
|
+
def amend(
|
|
168
|
+
self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
|
|
169
|
+
) -> int:
|
|
170
|
+
raise StoreError(_UNWRITABLE.format(engine=self.engine, verb="change"))
|
|
171
|
+
|
|
172
|
+
def remove(self, collection: str, key: str = "", *, by: str = "") -> int:
|
|
173
|
+
raise StoreError(_UNWRITABLE.format(engine=self.engine, verb="remove from"))
|
|
174
|
+
|
|
175
|
+
def clear(self) -> None:
|
|
176
|
+
"""Empty it, by restoring a snapshot that holds nothing."""
|
|
177
|
+
self.restore(Snapshot()) # type: ignore[attr-defined]
|
|
178
|
+
|
|
179
|
+
def save_to(self, path: str | Path) -> None:
|
|
180
|
+
import json
|
|
181
|
+
|
|
182
|
+
root = Path(path)
|
|
183
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
184
|
+
frozen = self.freeze() # type: ignore[attr-defined]
|
|
185
|
+
(root / SAVED).write_text(
|
|
186
|
+
json.dumps(
|
|
187
|
+
{
|
|
188
|
+
# The schema as the scripts that made it, because the rows alone cannot
|
|
189
|
+
# come back: a fresh engine has no tables to put them in.
|
|
190
|
+
"schema": list(getattr(self, "applied", [])),
|
|
191
|
+
"rows": frozen.rows,
|
|
192
|
+
"counters": frozen.counters,
|
|
193
|
+
},
|
|
194
|
+
indent=2,
|
|
195
|
+
default=str,
|
|
196
|
+
),
|
|
197
|
+
encoding="utf-8",
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
def load_from(self, path: str | Path) -> None:
|
|
201
|
+
import json
|
|
202
|
+
|
|
203
|
+
held = Path(path) / SAVED
|
|
204
|
+
if not held.exists():
|
|
205
|
+
raise StoreError(f"no saved store at {held}")
|
|
206
|
+
kept = json.loads(held.read_text(encoding="utf-8"))
|
|
207
|
+
for script in kept.get("schema") or []:
|
|
208
|
+
self.apply(script) # type: ignore[attr-defined]
|
|
209
|
+
self.restore(
|
|
210
|
+
Snapshot(rows=kept.get("rows") or {}, counters=kept.get("counters") or {})
|
|
211
|
+
) # type: ignore[attr-defined]
|
|
212
|
+
|
|
213
|
+
def close(self) -> None:
|
|
214
|
+
self.stop() # type: ignore[attr-defined]
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# What a saved container store is written as. Not the engine's own dump format: a snapshot is
|
|
218
|
+
# already engine-independent, and a dump would tie the saved world to the version that wrote it.
|
|
219
|
+
SAVED = "store.json"
|
|
220
|
+
|
|
221
|
+
_UNWRITABLE = (
|
|
222
|
+
"{engine} has no way to {verb} a collection one record at a time, so a scenario cannot set "
|
|
223
|
+
"up on it. Give the store add, amend and remove in this engine's own language."
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
_REGISTRY: dict[str, Callable[..., Store]] = {}
|
|
227
|
+
|
|
228
|
+
# Names people and manifests actually write, pointing at the engine they mean. Kept explicit
|
|
229
|
+
# rather than normalised in code, because guessing which engine an unrecognised word meant is
|
|
230
|
+
# how an agent ends up graded against the wrong one.
|
|
231
|
+
_ALIASES = {
|
|
232
|
+
"": "in_process",
|
|
233
|
+
"none": "in_process",
|
|
234
|
+
"memory": "in_process",
|
|
235
|
+
"in-memory": "in_process",
|
|
236
|
+
"inprocess": "in_process",
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def register_store(engine: str, factory: Callable[..., Store]) -> None:
|
|
241
|
+
"""Teach the harness an engine. A class and this line.
|
|
242
|
+
|
|
243
|
+
The cost of this line is what decides whether "whatever the agent uses" is real or an
|
|
244
|
+
aspiration, which is why the shared work lives in ``ContainerStore`` and an engine
|
|
245
|
+
contributes only what genuinely differs.
|
|
246
|
+
"""
|
|
247
|
+
_REGISTRY[engine] = factory
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def supported() -> tuple[str, ...]:
|
|
251
|
+
return tuple(sorted(_REGISTRY))
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def resolve(engine: str = "", **options: Any) -> Store:
|
|
255
|
+
"""The store for an engine, or a refusal naming what there is.
|
|
256
|
+
|
|
257
|
+
Deliberately not a fallback. An agent on an engine nobody has taught the harness to run is a
|
|
258
|
+
gap worth reporting, and quietly handing it a different store would produce a green suite
|
|
259
|
+
about queries the agent never executes.
|
|
260
|
+
"""
|
|
261
|
+
named = (engine or "").strip().lower()
|
|
262
|
+
named = _ALIASES.get(named, named)
|
|
263
|
+
if named not in _REGISTRY:
|
|
264
|
+
raise StoreError(
|
|
265
|
+
f"no store for engine {named!r}; the harness can stand up "
|
|
266
|
+
f"{', '.join(supported()) or 'nothing yet'}. Adding one is a class with the record "
|
|
267
|
+
"methods and a call to register_store, or write_store_ops for an engine in a container."
|
|
268
|
+
)
|
|
269
|
+
return _REGISTRY[named](**options)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
# The name the rest of the harness has always called this by.
|
|
273
|
+
open_store = resolve
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
from .inprocess import InProcessStore # noqa: E402
|
|
277
|
+
from .sqlite import SqliteStore # noqa: E402
|
|
278
|
+
|
|
279
|
+
register_store(SqliteStore.engine, SqliteStore)
|
|
280
|
+
register_store(InProcessStore.engine, InProcessStore)
|
|
281
|
+
|
|
282
|
+
from .container import ContainerStore, docker, strays # noqa: E402
|
|
283
|
+
from .postgres import PostgresStore # noqa: E402
|
|
284
|
+
|
|
285
|
+
# Postgres is registered as the worked example, not as the supported list. An engine the harness
|
|
286
|
+
# has never seen is meant to be written at build time against ``ContainerStore`` and proved by
|
|
287
|
+
# the gates, rather than waiting for someone to ship a class for it.
|
|
288
|
+
register_store(PostgresStore.engine, PostgresStore)
|
|
289
|
+
|
|
290
|
+
__all__ = [
|
|
291
|
+
"ContainerStore",
|
|
292
|
+
"InProcessStore",
|
|
293
|
+
"PostgresStore",
|
|
294
|
+
"Records",
|
|
295
|
+
"Snapshot",
|
|
296
|
+
"SqliteStore",
|
|
297
|
+
"Store",
|
|
298
|
+
"StoreError",
|
|
299
|
+
"docker",
|
|
300
|
+
"open_store",
|
|
301
|
+
"register_store",
|
|
302
|
+
"resolve",
|
|
303
|
+
"strays",
|
|
304
|
+
"supported",
|
|
305
|
+
]
|