agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""What a kind of world has to be able to do, so the checks do not care which kind it is.
|
|
2
|
+
|
|
3
|
+
A world backed by a database and a world backed by a page are different in every detail and the
|
|
4
|
+
same in what matters: something either exists in them or does not, an action either takes effect
|
|
5
|
+
or is refused, and what an action leaves behind is either carried or lost. Those are the things
|
|
6
|
+
worth checking, and none of them mention a table.
|
|
7
|
+
|
|
8
|
+
So the checks are written against this, and a kind supplies the four answers only it can give:
|
|
9
|
+
what exists, what the mutable state is, how to freeze it, and how to put it back. Adding a kind
|
|
10
|
+
is a class and a registration; nothing in the gate changes.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any, Callable, Mapping, Protocol, runtime_checkable
|
|
16
|
+
|
|
17
|
+
from .runtime import GeneratedWorld
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _rows(collection: Any) -> list[Any]:
|
|
21
|
+
"""One collection's members, whichever shape it is kept in.
|
|
22
|
+
|
|
23
|
+
A table gives a list of row mappings. A collection the agent's own code owns is as often a
|
|
24
|
+
mapping keyed by identifier, and iterating that yields keys rather than records, which is how
|
|
25
|
+
a check written for one shape silently reads the other.
|
|
26
|
+
"""
|
|
27
|
+
if isinstance(collection, dict):
|
|
28
|
+
return list(collection.values())
|
|
29
|
+
if isinstance(collection, (list, tuple)):
|
|
30
|
+
return list(collection)
|
|
31
|
+
return [collection]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _identifiers(state: Mapping[str, Any]) -> set[str]:
|
|
35
|
+
found: set[str] = set()
|
|
36
|
+
for name, collection in state.items():
|
|
37
|
+
if isinstance(collection, dict):
|
|
38
|
+
# The keys of a mapping are identifiers in their own right, and usually the ones a
|
|
39
|
+
# tool is called with.
|
|
40
|
+
found.update(str(key) for key in collection if isinstance(key, str) and key)
|
|
41
|
+
for row in _rows(collection):
|
|
42
|
+
if isinstance(row, Mapping):
|
|
43
|
+
found.update(
|
|
44
|
+
value for value in row.values() if isinstance(value, str) and value
|
|
45
|
+
)
|
|
46
|
+
elif isinstance(row, str) and row:
|
|
47
|
+
found.add(row)
|
|
48
|
+
return found
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _sizes(state: Mapping[str, Any]) -> dict[str, int]:
|
|
52
|
+
return {name: len(_rows(collection)) for name, collection in state.items()}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@runtime_checkable
|
|
56
|
+
class WorldKind(Protocol):
|
|
57
|
+
"""The per-kind half of a world. The shared half is ``GeneratedWorld``."""
|
|
58
|
+
|
|
59
|
+
key: str
|
|
60
|
+
label: str
|
|
61
|
+
|
|
62
|
+
def values_present(self, world: GeneratedWorld) -> set[str]:
|
|
63
|
+
"""Every identifier this world contains.
|
|
64
|
+
|
|
65
|
+
Answers whether the catalogue is complete: a contract that permits a value the world has
|
|
66
|
+
never heard of produces a tool that refuses forever, which is indistinguishable from a
|
|
67
|
+
tool being correctly strict.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
def mutable_state(self, world: GeneratedWorld) -> dict[str, int]:
|
|
71
|
+
"""Named parts of the world that an action can change, and how much is in each.
|
|
72
|
+
|
|
73
|
+
Used for two things: noticing that a saved world still holds whatever the builder was
|
|
74
|
+
experimenting with, and noticing that a sequence of actions left nothing behind.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
def describe(self, world: GeneratedWorld) -> str:
|
|
78
|
+
"""A short human-readable account of what is in the world."""
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class SqliteWorld:
|
|
82
|
+
"""A world whose state is rows in tables. Tool APIs, and anything with a data store."""
|
|
83
|
+
|
|
84
|
+
key = "sqlite"
|
|
85
|
+
label = "a database behind the agent's tools"
|
|
86
|
+
|
|
87
|
+
def values_present(self, world: GeneratedWorld) -> set[str]:
|
|
88
|
+
return _identifiers(world.state())
|
|
89
|
+
|
|
90
|
+
def mutable_state(self, world: GeneratedWorld) -> dict[str, int]:
|
|
91
|
+
return _sizes(world.state())
|
|
92
|
+
|
|
93
|
+
def describe(self, world: GeneratedWorld) -> str:
|
|
94
|
+
counts = self.mutable_state(world)
|
|
95
|
+
return ", ".join(f"{name}: {count}" for name, count in sorted(counts.items()))
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class BrowserWorld:
|
|
99
|
+
"""A world whose state is pages and the actions that change them.
|
|
100
|
+
|
|
101
|
+
ALK already carries a browser environment fed DOM snapshots and action fixtures, and it
|
|
102
|
+
already refuses a click matching no fixture. So this is the same move as the database kind:
|
|
103
|
+
generate instances of a shape that exists, rather than invent a mechanism.
|
|
104
|
+
|
|
105
|
+
What exists here is the set of things an agent can reach, which is selectors and URLs rather
|
|
106
|
+
than ids; what changes is which snapshot is current and what the actions have mutated.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
key = "browser"
|
|
110
|
+
label = "pages and the actions that change them"
|
|
111
|
+
|
|
112
|
+
def values_present(self, world: GeneratedWorld) -> set[str]:
|
|
113
|
+
present: set[str] = set()
|
|
114
|
+
for collection in world.state().values():
|
|
115
|
+
for row in _rows(collection):
|
|
116
|
+
if not isinstance(row, Mapping):
|
|
117
|
+
continue
|
|
118
|
+
for column in ("url", "selector", "id", "name", "action"):
|
|
119
|
+
value = row.get(column)
|
|
120
|
+
if isinstance(value, str) and value:
|
|
121
|
+
present.add(value)
|
|
122
|
+
return present
|
|
123
|
+
|
|
124
|
+
def mutable_state(self, world: GeneratedWorld) -> dict[str, int]:
|
|
125
|
+
return _sizes(world.state())
|
|
126
|
+
|
|
127
|
+
def describe(self, world: GeneratedWorld) -> str:
|
|
128
|
+
counts = self.mutable_state(world)
|
|
129
|
+
return ", ".join(f"{name}: {count}" for name, count in sorted(counts.items()))
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class InProcessWorld:
|
|
133
|
+
"""A world whose state the agent's own code holds, rather than a store we stood up.
|
|
134
|
+
|
|
135
|
+
This is what an adopted world usually is: the agent's tools were written to act on a structure
|
|
136
|
+
they build themselves, so the world holds that structure and does not interpret it.
|
|
137
|
+
"""
|
|
138
|
+
|
|
139
|
+
key = "in_process"
|
|
140
|
+
label = "state the agent's own code keeps"
|
|
141
|
+
|
|
142
|
+
def values_present(self, world: GeneratedWorld) -> set[str]:
|
|
143
|
+
return _identifiers(world.state())
|
|
144
|
+
|
|
145
|
+
def mutable_state(self, world: GeneratedWorld) -> dict[str, int]:
|
|
146
|
+
return _sizes(world.state())
|
|
147
|
+
|
|
148
|
+
def describe(self, world: GeneratedWorld) -> str:
|
|
149
|
+
counts = self.mutable_state(world)
|
|
150
|
+
return ", ".join(f"{name}: {count}" for name, count in sorted(counts.items()))
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
_REGISTRY: dict[str, Callable[[], WorldKind]] = {
|
|
154
|
+
SqliteWorld.key: SqliteWorld,
|
|
155
|
+
BrowserWorld.key: BrowserWorld,
|
|
156
|
+
InProcessWorld.key: InProcessWorld,
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def register_kind(key: str, factory: Callable[[], WorldKind]) -> None:
|
|
161
|
+
"""Add a kind of world. Computer use, a filesystem, a queue: a class and this line."""
|
|
162
|
+
_REGISTRY[key] = factory
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def resolve(key: str) -> WorldKind:
|
|
166
|
+
if key not in _REGISTRY:
|
|
167
|
+
raise NotImplementedError(
|
|
168
|
+
f"no world kind {key!r}; registered kinds are {', '.join(sorted(_REGISTRY))}"
|
|
169
|
+
)
|
|
170
|
+
return _REGISTRY[key]()
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def supported() -> tuple[str, ...]:
|
|
174
|
+
return tuple(sorted(_REGISTRY))
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def for_contract(contract: Any) -> WorldKind:
|
|
178
|
+
"""The kind of world an agent needs, from what the contract says it is.
|
|
179
|
+
|
|
180
|
+
Chosen rather than guessed at build time: an agent reachable by voice and by browser is one
|
|
181
|
+
agent with two runtimes, and which world to build is a decision about what is being tested.
|
|
182
|
+
"""
|
|
183
|
+
# What the store is, when the contract knows. That is the honest source: how a person reaches
|
|
184
|
+
# the agent says nothing about what its tools read and write, and an agent whose state lives
|
|
185
|
+
# in its own process is not a database however it is spoken to.
|
|
186
|
+
store = getattr(contract, "data_store", None)
|
|
187
|
+
named = str(getattr(store, "kind", "") or "").lower()
|
|
188
|
+
if named in _REGISTRY:
|
|
189
|
+
return resolve(named)
|
|
190
|
+
if named in ("in_process", "memory", "in-memory", "none", ""):
|
|
191
|
+
if named:
|
|
192
|
+
return resolve("in_process")
|
|
193
|
+
modality = str(getattr(contract, "modality", "") or "").lower()
|
|
194
|
+
if modality in ("browser", "computer_use", "cua"):
|
|
195
|
+
return resolve("browser")
|
|
196
|
+
return resolve("sqlite")
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Breaking a world on purpose, to find out whether its checks would notice.
|
|
2
|
+
|
|
3
|
+
The checks that verify an environment are written by whoever built it. That is the right way
|
|
4
|
+
round: what makes a world usable is a judgement about this agent, and no fixed set of probes
|
|
5
|
+
written in advance can make it for every agent. But it leaves nothing independent confirming the
|
|
6
|
+
checks work, and a check that cannot fail reports a healthy world forever.
|
|
7
|
+
|
|
8
|
+
So the checks are put to a test they cannot talk their way out of. The world is damaged in ways
|
|
9
|
+
that are obviously wrong, and the checks have to go red. One that stays green through every
|
|
10
|
+
damaged world is not verifying anything, whatever it claims to inspect.
|
|
11
|
+
|
|
12
|
+
The damage is deliberately generic, because a mutation that needed to understand the agent would
|
|
13
|
+
need the same judgement the checks needed, and nothing would be gained:
|
|
14
|
+
|
|
15
|
+
- **emptied**: every collection loses its contents. A world with no data at all.
|
|
16
|
+
- **silenced**: every tool answers with nothing. Calls succeed and change nothing.
|
|
17
|
+
|
|
18
|
+
Any check worth keeping fails against at least one of those. Most fail against both.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import copy
|
|
24
|
+
from typing import Any, Callable
|
|
25
|
+
|
|
26
|
+
from .runtime import GeneratedWorld
|
|
27
|
+
|
|
28
|
+
EMPTIED = "emptied"
|
|
29
|
+
# Not a kind of damage: how the report says the damage itself did not happen.
|
|
30
|
+
UNDAMAGED = "could not be damaged"
|
|
31
|
+
SILENCED = "silenced"
|
|
32
|
+
|
|
33
|
+
# What a silenced tool answers. Deliberately a plain empty string: it is the shape a handler
|
|
34
|
+
# returns when it has done nothing, which is exactly the failure being simulated.
|
|
35
|
+
_MUTE = "def handle(args, db):\n return ''\n"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _empty(world: GeneratedWorld) -> None:
|
|
39
|
+
"""Take the contents out of the world, leaving its shape intact.
|
|
40
|
+
|
|
41
|
+
The store empties itself where it can, because only it knows what its engine needs: a
|
|
42
|
+
relational one has to suspend foreign keys, or deleting a referenced table fails and most of
|
|
43
|
+
the data stays. Dropping collection by collection through the world's own vocabulary is the
|
|
44
|
+
fallback, for a store that has no opinion.
|
|
45
|
+
"""
|
|
46
|
+
emptied = getattr(world.store, "clear", None)
|
|
47
|
+
if callable(emptied):
|
|
48
|
+
emptied()
|
|
49
|
+
else:
|
|
50
|
+
for name in list(world.state()):
|
|
51
|
+
try:
|
|
52
|
+
world.drop(name)
|
|
53
|
+
except Exception:
|
|
54
|
+
# One collection that will not empty is not a reason to abandon the mutation. What
|
|
55
|
+
# survives is reported by `left`, so the gate can tell a check that failed to
|
|
56
|
+
# notice from a mutation that never happened.
|
|
57
|
+
continue
|
|
58
|
+
# The agent's own state, where its code keeps what its tools act on. A world can have both.
|
|
59
|
+
held = world.state_object
|
|
60
|
+
if isinstance(held, dict):
|
|
61
|
+
for name, group in held.items():
|
|
62
|
+
if isinstance(group, dict):
|
|
63
|
+
group.clear()
|
|
64
|
+
elif isinstance(group, list):
|
|
65
|
+
group.clear()
|
|
66
|
+
else:
|
|
67
|
+
held[name] = None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def left(world: GeneratedWorld) -> dict[str, int]:
|
|
71
|
+
"""What is still in the world after it was supposed to be empty.
|
|
72
|
+
|
|
73
|
+
The gate accuses a check of verifying nothing when it stays green through damage. That
|
|
74
|
+
accusation is only fair if the damage actually happened: a store that quietly refused to
|
|
75
|
+
empty leaves every check reading real data and looking vacuous, and the person then rewrites
|
|
76
|
+
a check that was right all along.
|
|
77
|
+
"""
|
|
78
|
+
return {name: len(rows) for name, rows in world.state().items() if rows}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _silence(world: GeneratedWorld) -> None:
|
|
82
|
+
"""Leave every tool answering with nothing, so no call has any effect."""
|
|
83
|
+
for name in list(world.handlers):
|
|
84
|
+
world.handlers[name] = _MUTE
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def damage() -> dict[str, Callable[[GeneratedWorld], None]]:
|
|
88
|
+
"""Every way a world is broken on purpose, by name."""
|
|
89
|
+
return {EMPTIED: _empty, SILENCED: _silence}
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def unnoticed(
|
|
93
|
+
world_root: Any,
|
|
94
|
+
checks: list[tuple[str, str]],
|
|
95
|
+
*,
|
|
96
|
+
run: Callable[[str, GeneratedWorld], Any],
|
|
97
|
+
restore: Callable[[Any], GeneratedWorld],
|
|
98
|
+
) -> dict[str, list[str]]:
|
|
99
|
+
"""Which checks fail to notice each kind of damage.
|
|
100
|
+
|
|
101
|
+
Every mutation runs against its own restored copy, so one cannot inherit another's damage and
|
|
102
|
+
a check is never blamed for a world some earlier mutation had already emptied.
|
|
103
|
+
|
|
104
|
+
Returns damage name to the checks that stayed green through it. A check appearing under every
|
|
105
|
+
kind of damage is one that cannot fail.
|
|
106
|
+
"""
|
|
107
|
+
survived: dict[str, list[str]] = {}
|
|
108
|
+
for name, apply in damage().items():
|
|
109
|
+
broken = restore(world_root)
|
|
110
|
+
try:
|
|
111
|
+
apply(broken)
|
|
112
|
+
if name == EMPTIED:
|
|
113
|
+
remaining = left(broken)
|
|
114
|
+
if remaining:
|
|
115
|
+
# The mutation did not land, so nothing can be concluded from it. Saying so is
|
|
116
|
+
# the point: reporting these checks as blind would have somebody rewrite a
|
|
117
|
+
# check that was reading the world correctly the whole time.
|
|
118
|
+
survived[name] = []
|
|
119
|
+
survived.setdefault(UNDAMAGED, []).append(
|
|
120
|
+
f"the world would not empty: {remaining}"
|
|
121
|
+
)
|
|
122
|
+
continue
|
|
123
|
+
still_green = []
|
|
124
|
+
for check_name, source in checks:
|
|
125
|
+
outcome = run(source, broken)
|
|
126
|
+
# A check that raises has not verified anything either, but that is a broken
|
|
127
|
+
# check rather than a blind one, and it is reported separately by the caller.
|
|
128
|
+
if getattr(outcome, "held", False):
|
|
129
|
+
still_green.append(check_name)
|
|
130
|
+
survived[name] = still_green
|
|
131
|
+
finally:
|
|
132
|
+
broken.close()
|
|
133
|
+
return survived
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def unnoticed_in_place(
|
|
137
|
+
world: GeneratedWorld,
|
|
138
|
+
checks: list[tuple[str, str]],
|
|
139
|
+
*,
|
|
140
|
+
run: Callable[[str, GeneratedWorld], Any],
|
|
141
|
+
) -> dict[str, list[str]]:
|
|
142
|
+
"""Mutation-test checks against a live service-backed world, restoring it each time.
|
|
143
|
+
|
|
144
|
+
A source-provisioned database is owned by the repository's Compose project. Serialising it
|
|
145
|
+
to a temporary generic world loses that ownership information and makes restore try to boot
|
|
146
|
+
a second database from generated DDL. The live store already supplies exact checkpoint and
|
|
147
|
+
restore operations, so use those and put both data and forwarded handlers back after every
|
|
148
|
+
mutation.
|
|
149
|
+
"""
|
|
150
|
+
baseline = world.checkpoint()
|
|
151
|
+
handlers = copy.deepcopy(world.handlers)
|
|
152
|
+
survived: dict[str, list[str]] = {}
|
|
153
|
+
for name, apply in damage().items():
|
|
154
|
+
try:
|
|
155
|
+
apply(world)
|
|
156
|
+
if name == EMPTIED:
|
|
157
|
+
remaining = left(world)
|
|
158
|
+
if remaining:
|
|
159
|
+
survived[name] = []
|
|
160
|
+
survived.setdefault(UNDAMAGED, []).append(
|
|
161
|
+
f"the world would not empty: {remaining}"
|
|
162
|
+
)
|
|
163
|
+
continue
|
|
164
|
+
survived[name] = [
|
|
165
|
+
check_name
|
|
166
|
+
for check_name, source in checks
|
|
167
|
+
if getattr(run(source, world), "held", False)
|
|
168
|
+
]
|
|
169
|
+
finally:
|
|
170
|
+
world.revert(baseline)
|
|
171
|
+
world.handlers = copy.deepcopy(handlers)
|
|
172
|
+
return survived
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def blind(survived: dict[str, list[str]]) -> list[str]:
|
|
176
|
+
"""Checks that stayed green through every kind of damage."""
|
|
177
|
+
if not survived:
|
|
178
|
+
return []
|
|
179
|
+
kinds = [names for kind, names in survived.items() if kind != UNDAMAGED]
|
|
180
|
+
if not kinds:
|
|
181
|
+
return []
|
|
182
|
+
return (
|
|
183
|
+
sorted(set(kinds[0]).intersection(*kinds[1:]))
|
|
184
|
+
if len(kinds) > 1
|
|
185
|
+
else sorted(kinds[0])
|
|
186
|
+
)
|