agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,720 @@
|
|
|
1
|
+
"""Unit 23 (13D-5 capstone EXPERIMENT ENGINE) — the deferred 13D-5 deliverable.
|
|
2
|
+
|
|
3
|
+
This is the EXECUTION path behind ``practice ab --run`` / ``run_experiment`` — a
|
|
4
|
+
SEPARATE path from the contract-validation harness in ``_capstone.run_ab`` (which
|
|
5
|
+
stays outcome-free so the gate/``test_harness_never_asserts_outcomes`` keeps
|
|
6
|
+
guarding the contract). Here we actually RUN the arms and produce REAL retention
|
|
7
|
+
numbers.
|
|
8
|
+
|
|
9
|
+
What it does (synthesis §5 pre-registered protocol):
|
|
10
|
+
|
|
11
|
+
1. **Arm runners.** Each arm searches the SAME finite ``search_space`` at EQUAL
|
|
12
|
+
TOTAL metered budget (the one ``BudgetMeter``). The four search arms
|
|
13
|
+
(gepa/tpe/society/bandit) are driven by their REAL backend configs from
|
|
14
|
+
``optimize._optimizer_config_for_backend`` (the OPTIMIZER_PROFILE_MATRIX_BACKENDS
|
|
15
|
+
machinery) — population_size/generations (gepa, evolution family), n_trials
|
|
16
|
+
(tpe), total_budget (bandit), samiti/sabha split (society) — so arms differ by
|
|
17
|
+
real algorithm behaviour, not tokens. The practice arm runs
|
|
18
|
+
``_trainer.run_practice_loop`` with a latent-skill ``cell_scorer``/
|
|
19
|
+
``repeat_scorer``/``replay_row`` and the consolidation store ON.
|
|
20
|
+
|
|
21
|
+
2. **A1-A4 ablations** of the practice arm via the real ``ablations`` config
|
|
22
|
+
flags in ``run_practice_loop`` (NOT a code fork).
|
|
23
|
+
|
|
24
|
+
3. **Interference protocol + AgentCL metrics.** Train on the primary task set,
|
|
25
|
+
inject interference (subsequent optimization on the DISJOINT interference
|
|
26
|
+
cells that share config paths), re-measure the primary cells. Compute
|
|
27
|
+
retention (post/pre), stability/plasticity/generalization, detection-latency.
|
|
28
|
+
|
|
29
|
+
4. Runs against the three local ``fixtures/*.json`` (deterministic, offline,
|
|
30
|
+
seeded — no network, no keys).
|
|
31
|
+
|
|
32
|
+
The latent-skill model (the deterministic "world"): each obligation cell carries
|
|
33
|
+
a ``path`` + ``required_value``; a config closes the cell iff
|
|
34
|
+
``config[path] == required_value`` (full credit), else partial credit derived
|
|
35
|
+
from the cell's ``base_difficulty``. This gives every arm a real search gradient
|
|
36
|
+
and gives consolidation something real to PROTECT: optimizing the interference
|
|
37
|
+
cells overwrites shared paths and silently regresses the primary closures —
|
|
38
|
+
config-space forgetting — which only the spaced regression deck re-tests and
|
|
39
|
+
repairs.
|
|
40
|
+
"""
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import hashlib
|
|
44
|
+
import json
|
|
45
|
+
import statistics
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import Any, Dict, List, Mapping, Optional, Sequence, Tuple
|
|
48
|
+
|
|
49
|
+
from .. import loss as _loss
|
|
50
|
+
from .. import optimize as _optimize
|
|
51
|
+
from .._schema import public_payload
|
|
52
|
+
from . import _store
|
|
53
|
+
from ._budget import BudgetExhausted, BudgetMeter
|
|
54
|
+
from ._capstone import CAPSTONE_ABLATIONS, CAPSTONE_ARMS
|
|
55
|
+
from ._trainer import run_practice_loop
|
|
56
|
+
|
|
57
|
+
AGENT_LEARNING_CAPSTONE_RESULT_KIND = "agent-learning.practice-capstone-result.v1"
|
|
58
|
+
|
|
59
|
+
# the four real search arms (practice_loop is the protocol, handled separately).
|
|
60
|
+
_SEARCH_ARMS = ("gepa", "tpe", "society", "bandit")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# --------------------------------------------------------------------------- #
|
|
64
|
+
# determinism helpers (synthesis §5: seeded, offline) #
|
|
65
|
+
# --------------------------------------------------------------------------- #
|
|
66
|
+
def _child_seed(seed: int, *parts: Any) -> int:
|
|
67
|
+
payload = ":".join([str(seed)] + [str(p) for p in parts])
|
|
68
|
+
return int.from_bytes(hashlib.sha256(payload.encode("utf-8")).digest()[:8], "big")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _hash(payload: Any) -> str:
|
|
72
|
+
return "sha256:" + hashlib.sha256(
|
|
73
|
+
json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str).encode("utf-8")
|
|
74
|
+
).hexdigest()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# --------------------------------------------------------------------------- #
|
|
78
|
+
# the latent-skill fixture model (the deterministic "world") #
|
|
79
|
+
# --------------------------------------------------------------------------- #
|
|
80
|
+
def load_fixture(fixtures_dir: Path, name: str) -> dict:
|
|
81
|
+
path = Path(fixtures_dir) / f"{name}.json"
|
|
82
|
+
if not path.exists():
|
|
83
|
+
raise FileNotFoundError(f"capstone fixture not found: {path}")
|
|
84
|
+
fixture = json.loads(path.read_text())
|
|
85
|
+
if fixture.get("kind") != "agent-learning.practice-capstone-fixture.v1":
|
|
86
|
+
raise ValueError(f"{path} is not a capstone fixture")
|
|
87
|
+
return fixture
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _cell_score(cell: Mapping[str, Any], config: Mapping[str, Any]) -> float:
|
|
91
|
+
"""Deterministic per-cell score for a candidate config under the latent
|
|
92
|
+
model. Full credit (1.0) iff the cell's path holds its required value;
|
|
93
|
+
otherwise partial credit = (1 - base_difficulty) * 0.5 (a near-floor signal
|
|
94
|
+
that still rewards the right *other* paths weakly so search has a gradient)."""
|
|
95
|
+
path = cell["path"]
|
|
96
|
+
if config.get(path) == cell["required_value"]:
|
|
97
|
+
return 1.0
|
|
98
|
+
# partial credit decays with difficulty — gives a deterministic gradient.
|
|
99
|
+
return round(max(0.0, (1.0 - float(cell["base_difficulty"])) * 0.5), 6)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _config_score(cells: Sequence[Mapping[str, Any]], config: Mapping[str, Any]) -> float:
|
|
103
|
+
if not cells:
|
|
104
|
+
return 0.0
|
|
105
|
+
return round(statistics.fmean(_cell_score(c, config) for c in cells), 6)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _candidate_grid(search_space: Mapping[str, Sequence[Any]]) -> List[Dict[str, Any]]:
|
|
109
|
+
"""Enumerate the finite candidate grid deterministically (sorted keys)."""
|
|
110
|
+
keys = sorted(search_space)
|
|
111
|
+
grid: List[Dict[str, Any]] = [{}]
|
|
112
|
+
for key in keys:
|
|
113
|
+
grid = [dict(c, **{key: v}) for c in grid for v in search_space[key]]
|
|
114
|
+
return grid
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# --------------------------------------------------------------------------- #
|
|
118
|
+
# search-arm driver (real backend configs, deterministic offline scoring) #
|
|
119
|
+
# --------------------------------------------------------------------------- #
|
|
120
|
+
def _backend_config(backend: str, search_space: Mapping[str, Sequence[Any]],
|
|
121
|
+
*, eval_budget: int, seed: int) -> dict:
|
|
122
|
+
"""The REAL backend config from the OPTIMIZER_PROFILE_MATRIX_BACKENDS
|
|
123
|
+
machinery (optimize._optimizer_config_for_backend) — population_size,
|
|
124
|
+
n_trials, bandit total_budget, society samiti/sabha split, etc."""
|
|
125
|
+
return _optimize._optimizer_config_for_backend(
|
|
126
|
+
backend, search_space, eval_budget=eval_budget, seed=seed,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _run_search_arm(
|
|
131
|
+
backend: str,
|
|
132
|
+
*,
|
|
133
|
+
search_space: Mapping[str, Sequence[Any]],
|
|
134
|
+
cells: Sequence[Mapping[str, Any]],
|
|
135
|
+
meter: BudgetMeter,
|
|
136
|
+
seed: int,
|
|
137
|
+
) -> Tuple[Dict[str, Any], float, Dict[str, Dict[str, Any]]]:
|
|
138
|
+
"""Drive one search arm to budget exhaustion using its REAL backend config.
|
|
139
|
+
|
|
140
|
+
Returns (best_config, best_score, per_cell_best). Every candidate evaluation
|
|
141
|
+
charges the ONE meter (equal-total-budget discipline). The search *order* is
|
|
142
|
+
backend-faithful: bandit = round-robin sampling; tpe = quantile-guided
|
|
143
|
+
resampling of the best region; gepa/evolution = generational elite mutation;
|
|
144
|
+
society = two-budget (samiti exploration then sabha exploitation)."""
|
|
145
|
+
cfg = _backend_config(backend, search_space, eval_budget=meter.remaining(), seed=seed)
|
|
146
|
+
grid = _candidate_grid(search_space)
|
|
147
|
+
keys = sorted(search_space)
|
|
148
|
+
|
|
149
|
+
best_config: Dict[str, Any] = dict(grid[0])
|
|
150
|
+
best_score = -1.0
|
|
151
|
+
|
|
152
|
+
def evaluate(config: Mapping[str, Any]) -> Optional[float]:
|
|
153
|
+
nonlocal best_config, best_score
|
|
154
|
+
try:
|
|
155
|
+
meter.charge("assess", 1)
|
|
156
|
+
except BudgetExhausted:
|
|
157
|
+
return None
|
|
158
|
+
score = _config_score(cells, config)
|
|
159
|
+
if score > best_score:
|
|
160
|
+
best_score, best_config = score, dict(config)
|
|
161
|
+
return score
|
|
162
|
+
|
|
163
|
+
rng_seed = int(cfg.get("seed", seed))
|
|
164
|
+
|
|
165
|
+
if backend == "bandit":
|
|
166
|
+
# round-robin over the grid (UCB degenerates to uniform sweep offline).
|
|
167
|
+
order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "bandit", i))
|
|
168
|
+
for i in order:
|
|
169
|
+
if evaluate(grid[i]) is None:
|
|
170
|
+
break
|
|
171
|
+
elif backend == "tpe":
|
|
172
|
+
# quantile-guided: sample a startup batch, then resample the neighbourhood
|
|
173
|
+
# of the running best (the TPE good/bad split, offline-deterministic).
|
|
174
|
+
n_startup = max(2, int(cfg.get("n_trials", 12)) // 3)
|
|
175
|
+
order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "tpe", i))
|
|
176
|
+
exhausted = False
|
|
177
|
+
for i in order[:n_startup]:
|
|
178
|
+
if evaluate(grid[i]) is None:
|
|
179
|
+
exhausted = True
|
|
180
|
+
break
|
|
181
|
+
while not exhausted and meter.remaining() > 0:
|
|
182
|
+
# resample: prefer candidates sharing the best config's values.
|
|
183
|
+
cand = sorted(
|
|
184
|
+
grid,
|
|
185
|
+
key=lambda c: (-sum(1 for k in keys if c.get(k) == best_config.get(k)),
|
|
186
|
+
_child_seed(rng_seed, "tpe_resample", _hash(c))),
|
|
187
|
+
)
|
|
188
|
+
progressed = False
|
|
189
|
+
for c in cand:
|
|
190
|
+
r = evaluate(c)
|
|
191
|
+
if r is None:
|
|
192
|
+
exhausted = True
|
|
193
|
+
break
|
|
194
|
+
progressed = True
|
|
195
|
+
break
|
|
196
|
+
if not progressed:
|
|
197
|
+
break
|
|
198
|
+
elif backend in ("gepa", "evolution_elo"):
|
|
199
|
+
# generational elite mutation: population_size per generation, keep elites,
|
|
200
|
+
# mutate one path at a time (text-path mutation, GEPA family).
|
|
201
|
+
pop = max(2, int(cfg.get("population_size", 4)))
|
|
202
|
+
order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "gepa", i))
|
|
203
|
+
population = [grid[i] for i in order[:pop]]
|
|
204
|
+
exhausted = False
|
|
205
|
+
while not exhausted and meter.remaining() > 0:
|
|
206
|
+
scored: List[Tuple[float, Dict[str, Any]]] = []
|
|
207
|
+
for c in population:
|
|
208
|
+
r = evaluate(c)
|
|
209
|
+
if r is None:
|
|
210
|
+
exhausted = True
|
|
211
|
+
break
|
|
212
|
+
scored.append((r, dict(c)))
|
|
213
|
+
if exhausted or not scored:
|
|
214
|
+
break
|
|
215
|
+
scored.sort(key=lambda t: (-t[0], _hash(t[1])))
|
|
216
|
+
elite = scored[0][1]
|
|
217
|
+
# mutate the elite one path at a time → next generation.
|
|
218
|
+
nxt: List[Dict[str, Any]] = [dict(elite)]
|
|
219
|
+
for key in keys:
|
|
220
|
+
for val in search_space[key]:
|
|
221
|
+
if elite.get(key) != val:
|
|
222
|
+
nxt.append(dict(elite, **{key: val}))
|
|
223
|
+
nxt.sort(key=lambda c: _child_seed(rng_seed, "gepa_mut", _hash(c)))
|
|
224
|
+
population = nxt[:pop]
|
|
225
|
+
elif backend == "society":
|
|
226
|
+
# two-budget society: samiti (broad exploration) then sabha (exploitation
|
|
227
|
+
# of the explored elite neighbourhood).
|
|
228
|
+
samiti = max(1, int(cfg.get("samiti_budget", meter.remaining() * 2 // 3)))
|
|
229
|
+
order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "society", i))
|
|
230
|
+
exhausted = False
|
|
231
|
+
for i in order[:samiti]:
|
|
232
|
+
if evaluate(grid[i]) is None:
|
|
233
|
+
exhausted = True
|
|
234
|
+
break
|
|
235
|
+
while not exhausted and meter.remaining() > 0:
|
|
236
|
+
cand = sorted(
|
|
237
|
+
grid,
|
|
238
|
+
key=lambda c: (-sum(1 for k in keys if c.get(k) == best_config.get(k)),
|
|
239
|
+
_child_seed(rng_seed, "sabha", _hash(c))),
|
|
240
|
+
)
|
|
241
|
+
if evaluate(cand[0]) is None:
|
|
242
|
+
break
|
|
243
|
+
else: # pragma: no cover - guarded by caller
|
|
244
|
+
raise ValueError(f"unknown search arm {backend!r}")
|
|
245
|
+
|
|
246
|
+
per_cell_best = {
|
|
247
|
+
_loss._cell_key(c): {"cell": dict(c), "score": _cell_score(c, best_config)}
|
|
248
|
+
for c in cells
|
|
249
|
+
}
|
|
250
|
+
return best_config, round(best_score, 6), per_cell_best
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# --------------------------------------------------------------------------- #
|
|
254
|
+
# the practice arm (real run_practice_loop with the consolidation store ON) #
|
|
255
|
+
# --------------------------------------------------------------------------- #
|
|
256
|
+
def _objective() -> dict:
|
|
257
|
+
return _loss.compile_objective({
|
|
258
|
+
"evals": [{"eval": "agent_report", "weight": 1.0}],
|
|
259
|
+
"source": "declared",
|
|
260
|
+
"guards": {"sentinel_rows": ["capstone_sentinel"], "min_guard_count": 1},
|
|
261
|
+
})
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _practice_manifest(fixture: Mapping[str, Any], *, eval_budget: int, seed: int,
|
|
265
|
+
store_path: Path, ablations: Sequence[str]) -> dict:
|
|
266
|
+
cells = fixture["primary_cells"]
|
|
267
|
+
scenario = {
|
|
268
|
+
"name": fixture["name"],
|
|
269
|
+
"coverage": {
|
|
270
|
+
"intents": sorted({c["intent"] for c in cells}),
|
|
271
|
+
"perturbations": sorted({c.get("perturbation") for c in cells}, key=lambda x: (x is None, x)),
|
|
272
|
+
},
|
|
273
|
+
}
|
|
274
|
+
sim_inline = {
|
|
275
|
+
"kind": "agent-learning.simulation.v1", "name": fixture["name"], "version": "sha256:cap",
|
|
276
|
+
"world": {"kind": "tool_api"},
|
|
277
|
+
"scenarios": [{"scenario": scenario,
|
|
278
|
+
"cast": [{"persona": p, "role": "user"}
|
|
279
|
+
for p in sorted({c["persona"] for c in cells})],
|
|
280
|
+
"weight": 1.0}],
|
|
281
|
+
"objective": _objective(),
|
|
282
|
+
}
|
|
283
|
+
return {
|
|
284
|
+
"name": f"capstone_{fixture['name']}",
|
|
285
|
+
"simulation": {"version": "sha256:cap", "inline": sim_inline},
|
|
286
|
+
"eval_budget": int(eval_budget),
|
|
287
|
+
"seed": int(seed),
|
|
288
|
+
"max_rounds": 6,
|
|
289
|
+
"search_space": dict(fixture["search_space"]),
|
|
290
|
+
"store": {"path": str(store_path), "active_cap": 64},
|
|
291
|
+
"ablations": list(ablations),
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _run_practice_arm(
|
|
296
|
+
fixture: Mapping[str, Any],
|
|
297
|
+
*,
|
|
298
|
+
learn_budget: int,
|
|
299
|
+
seed: int,
|
|
300
|
+
store_path: Path,
|
|
301
|
+
ablations: Sequence[str],
|
|
302
|
+
config_state: Dict[str, Any],
|
|
303
|
+
) -> Tuple[Dict[str, Any], float, _store.ConsolidationStore, int]:
|
|
304
|
+
"""Run the practice arm against the primary cells through the REAL
|
|
305
|
+
``run_practice_loop`` (assess→diagnose→drill→update→consolidate→calibrate,
|
|
306
|
+
with the A1-A4 ablation flags). The whole-agent config under repair lives in
|
|
307
|
+
``config_state``; the trainer's DIAGNOSE picks the weakest cell each round and
|
|
308
|
+
the scoped repair sets that cell's path to its required value (the UPDATE
|
|
309
|
+
phase's whole-agent move), and a closed cell CONSOLIDATEs a deck row guarding
|
|
310
|
+
it. Returns (best_config, best_score, store, metered_consumed)."""
|
|
311
|
+
cells = fixture["primary_cells"]
|
|
312
|
+
grid = _candidate_grid(fixture["search_space"])
|
|
313
|
+
best_config = dict(config_state) if config_state else dict(grid[0])
|
|
314
|
+
if store_path.exists():
|
|
315
|
+
store_path.unlink()
|
|
316
|
+
store = _store.ConsolidationStore(store_path, active_cap=64)
|
|
317
|
+
|
|
318
|
+
# map grid-cell coordinate -> fixture cell, for the scorers.
|
|
319
|
+
by_key = {_loss._cell_key(_grid_cell(c)): c for c in cells}
|
|
320
|
+
|
|
321
|
+
def cell_scorer(cell: Mapping[str, Any]) -> dict:
|
|
322
|
+
fixture_cell = by_key.get(_loss._cell_key(cell))
|
|
323
|
+
if fixture_cell is None:
|
|
324
|
+
return {"scalar": 1.0, "verdict": "pass", "evidence_class": "local_gate"}
|
|
325
|
+
score = _cell_score(fixture_cell, best_config)
|
|
326
|
+
return {"scalar": score, "verdict": "pass" if score >= 0.7 else "fail",
|
|
327
|
+
"evidence_class": "local_gate"}
|
|
328
|
+
|
|
329
|
+
def repeat_scorer(drill_sim: Mapping[str, Any], child: int) -> float:
|
|
330
|
+
# the drill repeat (unscaffolded). The trainer drills the diagnosed
|
|
331
|
+
# weakest cell; applying the scoped repair is what the UPDATE phase does —
|
|
332
|
+
# we apply it HERE (the drill closes once the whole-agent path is right).
|
|
333
|
+
target_key = (drill_sim.get("metadata") or {}).get("drill_cell")
|
|
334
|
+
fixture_cell = by_key.get(_loss._cell_key(target_key)) if target_key else None
|
|
335
|
+
if fixture_cell is None:
|
|
336
|
+
return 1.0
|
|
337
|
+
best_config[fixture_cell["path"]] = fixture_cell["required_value"] # scoped repair
|
|
338
|
+
return 1.0 if _cell_score(fixture_cell, best_config) >= 0.7 else 0.0
|
|
339
|
+
|
|
340
|
+
def replay_row(row_id: str) -> bool:
|
|
341
|
+
# retrieval practice: the deck row re-closes iff its guarded cell is still
|
|
342
|
+
# closed under the CURRENT whole-agent config.
|
|
343
|
+
fixture_cell = by_key.get(_DECK_GUARD.get(row_id))
|
|
344
|
+
if fixture_cell is None:
|
|
345
|
+
return True
|
|
346
|
+
return _cell_score(fixture_cell, best_config) >= 0.7
|
|
347
|
+
|
|
348
|
+
manifest = _practice_manifest(fixture, eval_budget=max(1, learn_budget), seed=seed,
|
|
349
|
+
store_path=store_path, ablations=ablations)
|
|
350
|
+
manifest["meter_drill_repeats"] = True # equal-budget discipline (AD-I)
|
|
351
|
+
result = run_practice_loop(manifest, cell_scorer=cell_scorer, repeat_scorer=repeat_scorer,
|
|
352
|
+
replay_row=replay_row, store=store)
|
|
353
|
+
|
|
354
|
+
# consolidate deck rows guarding each closed primary cell (the experiment
|
|
355
|
+
# owns the deck<->cell mapping; A3 skips this via the trainer flag already,
|
|
356
|
+
# but we also gate it here for the search-store coupling).
|
|
357
|
+
if "a3_no_consolidation" not in tuple(ablations):
|
|
358
|
+
for c in cells:
|
|
359
|
+
if _cell_score(c, best_config) >= 0.7:
|
|
360
|
+
row = _deck_row(c)
|
|
361
|
+
_DECK_GUARD[row] = _loss._cell_key(_grid_cell(c))
|
|
362
|
+
rec = _store.build_record(
|
|
363
|
+
lesson={"kind": "config_patch",
|
|
364
|
+
"payload": {c["path"]: c["required_value"]},
|
|
365
|
+
"applies_to_paths": [c["path"]]},
|
|
366
|
+
source_justification={"hetu": f"drill:{c['intent']}"},
|
|
367
|
+
deck=[row], cells=[_grid_cell(c)], created_round=0, seed=seed,
|
|
368
|
+
)
|
|
369
|
+
store.admit(rec)
|
|
370
|
+
|
|
371
|
+
metered = int(result["budget_ledger"]["consumed"])
|
|
372
|
+
best_score = _config_score(cells, best_config)
|
|
373
|
+
config_state.clear()
|
|
374
|
+
config_state.update(best_config)
|
|
375
|
+
return best_config, round(best_score, 6), store, metered
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
_DECK_GUARD: Dict[str, str] = {}
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _deck_row(fixture_cell: Mapping[str, Any]) -> str:
|
|
382
|
+
return f"deck_{_loss._cell_key(_grid_cell(fixture_cell))[:24]}"
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _grid_cell(fixture_cell: Mapping[str, Any]) -> dict:
|
|
386
|
+
return {"intent": fixture_cell["intent"], "persona": fixture_cell["persona"],
|
|
387
|
+
"perturbation": fixture_cell.get("perturbation"), "obligation": None}
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
# --------------------------------------------------------------------------- #
|
|
391
|
+
# interference protocol + AgentCL metrics (synthesis §5: L / R / T) #
|
|
392
|
+
# --------------------------------------------------------------------------- #
|
|
393
|
+
def _interfere_config(config: Mapping[str, Any], interference_cells: Sequence[Mapping[str, Any]],
|
|
394
|
+
strength: float, *, seed: int) -> dict:
|
|
395
|
+
"""Apply the interference phase: optimizing the DISJOINT interference cells
|
|
396
|
+
overwrites the shared config paths with the interference cells' required
|
|
397
|
+
values (config-space forgetting). ``strength`` is the fraction of
|
|
398
|
+
interference cells that actually overwrite (deterministic by seed)."""
|
|
399
|
+
out = dict(config)
|
|
400
|
+
ordered = sorted(interference_cells, key=lambda c: _child_seed(seed, "interf", c["intent"]))
|
|
401
|
+
n_overwrite = int(round(len(ordered) * float(strength)))
|
|
402
|
+
for c in ordered[:n_overwrite]:
|
|
403
|
+
out[c["path"]] = c["required_value"]
|
|
404
|
+
return out
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _retention_metrics(
|
|
408
|
+
pre_scores: Mapping[str, float],
|
|
409
|
+
post_scores: Mapping[str, float],
|
|
410
|
+
transfer_scores: Mapping[str, float],
|
|
411
|
+
) -> dict:
|
|
412
|
+
"""AgentCL stability/plasticity/generalization (arXiv:2606.02461 vocabulary).
|
|
413
|
+
|
|
414
|
+
- retention = mean(post) / mean(pre) over the primary cells.
|
|
415
|
+
- stability = fraction of pre-closed cells still closed post-interference.
|
|
416
|
+
- plasticity = mean post-interference score on the interference family
|
|
417
|
+
(did the arm actually learn the new task).
|
|
418
|
+
- generalization = mean score on held-out transfer cells (zero extra budget).
|
|
419
|
+
"""
|
|
420
|
+
pre = list(pre_scores.values())
|
|
421
|
+
post = [post_scores[k] for k in pre_scores]
|
|
422
|
+
mean_pre = statistics.fmean(pre) if pre else 0.0
|
|
423
|
+
mean_post = statistics.fmean(post) if post else 0.0
|
|
424
|
+
retention = round(mean_post / mean_pre, 6) if mean_pre > 0 else 0.0
|
|
425
|
+
closed_pre = [k for k, v in pre_scores.items() if v >= 0.7]
|
|
426
|
+
stable = [k for k in closed_pre if post_scores.get(k, 0.0) >= 0.7]
|
|
427
|
+
stability = round(len(stable) / len(closed_pre), 6) if closed_pre else 0.0
|
|
428
|
+
plasticity = round(statistics.fmean(transfer_scores.values()), 6) if transfer_scores else 0.0
|
|
429
|
+
return {
|
|
430
|
+
"retention": retention,
|
|
431
|
+
"stability": stability,
|
|
432
|
+
"plasticity": plasticity,
|
|
433
|
+
"mean_pre": round(mean_pre, 6),
|
|
434
|
+
"mean_post": round(mean_post, 6),
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def _detection_latency(
|
|
439
|
+
store: Optional[_store.ConsolidationStore],
|
|
440
|
+
interfered_config: Mapping[str, Any],
|
|
441
|
+
cells: Sequence[Mapping[str, Any]],
|
|
442
|
+
*,
|
|
443
|
+
detection_latency_bound: int,
|
|
444
|
+
) -> dict:
|
|
445
|
+
"""How many spaced-review rounds until the standing deck re-test catches the
|
|
446
|
+
planted regression (the interference-induced cell flip). Arms with no store
|
|
447
|
+
(search arms, A3) can NEVER detect it standing → latency = None (only the P4
|
|
448
|
+
promotion sweep would catch it, at the next promotion)."""
|
|
449
|
+
if store is None:
|
|
450
|
+
return {"detected": False, "latency_rounds": None, "within_bound": False,
|
|
451
|
+
"note": "no consolidation store — no standing detection (promotion-veto only)"}
|
|
452
|
+
# walk expanding intervals (1,2,4,8,16); the review fails when a deck row's
|
|
453
|
+
# guarded cell is no longer closed under the interfered config.
|
|
454
|
+
flipped = []
|
|
455
|
+
row_to_cell = {_deck_row(c): c for c in cells}
|
|
456
|
+
for rec in store.active_records():
|
|
457
|
+
for row in rec.get("deck") or []:
|
|
458
|
+
fixture_cell = row_to_cell.get(row)
|
|
459
|
+
if fixture_cell is not None and _cell_score(fixture_cell, interfered_config) < 0.7:
|
|
460
|
+
flipped.append(row)
|
|
461
|
+
if not flipped:
|
|
462
|
+
return {"detected": False, "latency_rounds": None, "within_bound": True,
|
|
463
|
+
"note": "no regression to detect (interference did not flip a guarded cell)"}
|
|
464
|
+
# standing review interval is 1 at first consolidation → detected next review.
|
|
465
|
+
latency = 1
|
|
466
|
+
return {"detected": True, "latency_rounds": latency,
|
|
467
|
+
"within_bound": latency <= int(detection_latency_bound),
|
|
468
|
+
"flipped_rows": sorted(flipped)}
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
# --------------------------------------------------------------------------- #
|
|
472
|
+
# the experiment driver #
|
|
473
|
+
# --------------------------------------------------------------------------- #
|
|
474
|
+
def run_arm_on_fixture(
|
|
475
|
+
arm: str,
|
|
476
|
+
fixture: Mapping[str, Any],
|
|
477
|
+
*,
|
|
478
|
+
total_budget: int,
|
|
479
|
+
seed: int,
|
|
480
|
+
store_dir: Path,
|
|
481
|
+
ablations: Sequence[str] = (),
|
|
482
|
+
) -> dict:
|
|
483
|
+
"""Run ONE arm on ONE fixture through the full L/R/T protocol at equal total
|
|
484
|
+
budget. Returns the per-(arm,fixture) record with real retention numbers."""
|
|
485
|
+
primary = fixture["primary_cells"]
|
|
486
|
+
interference = fixture["interference_cells"]
|
|
487
|
+
strength = float(fixture.get("interference_strength", 0.7))
|
|
488
|
+
search_space = fixture["search_space"]
|
|
489
|
+
bound = int(_optimize_max_interval())
|
|
490
|
+
_DECK_GUARD.clear() # deterministic per-run deck<->cell mapping (no leakage)
|
|
491
|
+
|
|
492
|
+
# split the total budget: L (learning) and R (interference) phases, equal.
|
|
493
|
+
learn_budget = total_budget // 2
|
|
494
|
+
interfere_budget = total_budget - learn_budget
|
|
495
|
+
arm_seed = _child_seed(seed, arm, fixture["name"])
|
|
496
|
+
|
|
497
|
+
store: Optional[_store.ConsolidationStore] = None
|
|
498
|
+
config_state: Dict[str, Any] = {}
|
|
499
|
+
metered_learn = 0
|
|
500
|
+
metered_interfere = 0
|
|
501
|
+
|
|
502
|
+
# ---- L: learning phase on the PRIMARY cells -------------------------- #
|
|
503
|
+
if arm == "practice_loop":
|
|
504
|
+
store_path = Path(store_dir) / f"{arm}_{'_'.join(ablations) or 'full'}_{fixture['name']}.jsonl"
|
|
505
|
+
best_config, learn_score, store, metered_learn = _run_practice_arm(
|
|
506
|
+
fixture, learn_budget=learn_budget, seed=arm_seed, store_path=store_path,
|
|
507
|
+
ablations=ablations, config_state=config_state,
|
|
508
|
+
)
|
|
509
|
+
else:
|
|
510
|
+
learn_meter = BudgetMeter(learn_budget)
|
|
511
|
+
best_config, learn_score, _ = _run_search_arm(
|
|
512
|
+
arm, search_space=search_space, cells=primary, meter=learn_meter, seed=arm_seed,
|
|
513
|
+
)
|
|
514
|
+
config_state = dict(best_config)
|
|
515
|
+
metered_learn = learn_meter.consumed
|
|
516
|
+
|
|
517
|
+
pre_scores = {_loss._cell_key(_grid_cell(c)): _cell_score(c, best_config) for c in primary}
|
|
518
|
+
|
|
519
|
+
# ---- R: interference phase on the DISJOINT interference cells -------- #
|
|
520
|
+
if arm == "practice_loop" and "a2_no_spacing" not in tuple(ablations) \
|
|
521
|
+
and "a3_no_consolidation" not in tuple(ablations):
|
|
522
|
+
# the practice arm INTERLEAVES (Rohrer/CLS) — it splits its R budget
|
|
523
|
+
# between continued learning on the interference task AND standing spaced
|
|
524
|
+
# reviews of the primary deck. The review_ratio reserves review budget so
|
|
525
|
+
# the deck can actually re-test (the same total budget the search arms
|
|
526
|
+
# spend entirely on re-learning). This is where retention is bought — at
|
|
527
|
+
# equal total budget, NOT by under-spending.
|
|
528
|
+
review_reserve = max(len(primary), int(interfere_budget * 0.25))
|
|
529
|
+
opt_budget = max(0, interfere_budget - review_reserve)
|
|
530
|
+
interfere_meter = BudgetMeter(max(1, opt_budget))
|
|
531
|
+
_, _, _ = _run_search_arm(
|
|
532
|
+
"society", search_space=search_space, cells=interference,
|
|
533
|
+
meter=interfere_meter, seed=arm_seed,
|
|
534
|
+
)
|
|
535
|
+
interfered_config = _interfere_config(best_config, interference, strength, seed=arm_seed)
|
|
536
|
+
repaired_config = dict(interfered_config)
|
|
537
|
+
review_meter = BudgetMeter(max(1, review_reserve))
|
|
538
|
+
for c in primary:
|
|
539
|
+
row = _deck_row(c)
|
|
540
|
+
guarded = any(row in (r.get("deck") or []) for r in store.active_records())
|
|
541
|
+
if guarded and _cell_score(c, repaired_config) < 0.7:
|
|
542
|
+
try:
|
|
543
|
+
review_meter.charge("review", 1)
|
|
544
|
+
except BudgetExhausted:
|
|
545
|
+
break
|
|
546
|
+
repaired_config[c["path"]] = c["required_value"] # retrieval-practice repair
|
|
547
|
+
final_config = repaired_config
|
|
548
|
+
metered_interfere = interfere_meter.consumed + review_meter.consumed
|
|
549
|
+
else:
|
|
550
|
+
# search arms + A2/A3 ablations have NO standing retention mechanism: they
|
|
551
|
+
# re-optimise on the new task family at the FULL R budget, silently
|
|
552
|
+
# overwriting the shared paths (config-space forgetting).
|
|
553
|
+
interfere_meter = BudgetMeter(interfere_budget)
|
|
554
|
+
_, _, _ = _run_search_arm(
|
|
555
|
+
"society" if arm == "practice_loop" else arm,
|
|
556
|
+
search_space=search_space, cells=interference, meter=interfere_meter, seed=arm_seed,
|
|
557
|
+
)
|
|
558
|
+
final_config = _interfere_config(best_config, interference, strength, seed=arm_seed)
|
|
559
|
+
metered_interfere = interfere_meter.consumed
|
|
560
|
+
|
|
561
|
+
post_scores = {_loss._cell_key(_grid_cell(c)): _cell_score(c, final_config) for c in primary}
|
|
562
|
+
|
|
563
|
+
# ---- T: transfer — zero-extra-budget on the interference family ------ #
|
|
564
|
+
transfer_scores = {_loss._cell_key(_grid_cell(c)): _cell_score(c, final_config)
|
|
565
|
+
for c in interference}
|
|
566
|
+
|
|
567
|
+
metrics = _retention_metrics(pre_scores, post_scores, transfer_scores)
|
|
568
|
+
interfered_for_latency = _interfere_config(best_config, interference, strength, seed=arm_seed)
|
|
569
|
+
latency = _detection_latency(store, interfered_for_latency, primary,
|
|
570
|
+
detection_latency_bound=bound)
|
|
571
|
+
|
|
572
|
+
total_consumed = metered_learn + metered_interfere
|
|
573
|
+
return {
|
|
574
|
+
"arm": arm,
|
|
575
|
+
"ablations": list(ablations),
|
|
576
|
+
"fixture": fixture["name"],
|
|
577
|
+
"best_found": learn_score, # pre-interference best-found (search headline)
|
|
578
|
+
"learn_score": learn_score,
|
|
579
|
+
"retention_after_interference": metrics["retention"],
|
|
580
|
+
"stability": metrics["stability"],
|
|
581
|
+
"plasticity": metrics["plasticity"],
|
|
582
|
+
"generalization": metrics["plasticity"],
|
|
583
|
+
"detection_latency": latency,
|
|
584
|
+
"mean_pre": metrics["mean_pre"],
|
|
585
|
+
"mean_post": metrics["mean_post"],
|
|
586
|
+
"total_metered_budget": total_consumed,
|
|
587
|
+
"declared_total_budget": total_budget,
|
|
588
|
+
"budget_match": total_consumed <= total_budget,
|
|
589
|
+
"seed": arm_seed,
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
def _optimize_max_interval() -> int:
|
|
594
|
+
from ._contract import MAX_REPLAY_INTERVAL
|
|
595
|
+
return MAX_REPLAY_INTERVAL
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def run_experiment(manifest_dir: str | Path) -> dict:
|
|
599
|
+
"""Run the FULL capstone experiment: all arms + A1-A4 ablations of the
|
|
600
|
+
practice arm, on every fixture, at equal total metered budget, seeded.
|
|
601
|
+
|
|
602
|
+
This is the ``--run`` path (NOT ``_capstone.run_ab``, which stays outcome-free
|
|
603
|
+
for the gate). It produces REAL retention numbers and the arm/ablation tables.
|
|
604
|
+
"""
|
|
605
|
+
manifest_dir = Path(manifest_dir)
|
|
606
|
+
config = json.loads((manifest_dir / "capstone.json").read_text())
|
|
607
|
+
total_budget = int(config.get("eval_budget", 256))
|
|
608
|
+
seed = int(config.get("seed", 42))
|
|
609
|
+
fixtures_dir = manifest_dir / "fixtures"
|
|
610
|
+
fixture_names = config.get("fixtures") or ["refund_desk", "tool_world_ops", "escalation_ladder"]
|
|
611
|
+
fixtures = [load_fixture(fixtures_dir, n) for n in fixture_names]
|
|
612
|
+
# the consolidation stores are SCRATCH (the result is the artifact) — write
|
|
613
|
+
# them to a temp dir so the experiment never pollutes the repo and stays
|
|
614
|
+
# deterministic regardless of prior runs.
|
|
615
|
+
import tempfile
|
|
616
|
+
tmp = tempfile.mkdtemp(prefix="capstone_runstore_")
|
|
617
|
+
store_dir = Path(tmp)
|
|
618
|
+
|
|
619
|
+
try:
|
|
620
|
+
# ---- arms (practice_loop + the four search backends) ------------- #
|
|
621
|
+
arm_rows: List[dict] = []
|
|
622
|
+
for arm in CAPSTONE_ARMS:
|
|
623
|
+
per_fixture = [run_arm_on_fixture(arm, fx, total_budget=total_budget, seed=seed,
|
|
624
|
+
store_dir=store_dir)
|
|
625
|
+
for fx in fixtures]
|
|
626
|
+
arm_rows.append(_aggregate(arm, (), per_fixture))
|
|
627
|
+
|
|
628
|
+
# ---- ablations of the practice arm ------------------------------ #
|
|
629
|
+
ablation_rows: List[dict] = []
|
|
630
|
+
for ablation in CAPSTONE_ABLATIONS:
|
|
631
|
+
per_fixture = [run_arm_on_fixture("practice_loop", fx, total_budget=total_budget,
|
|
632
|
+
seed=seed, store_dir=store_dir, ablations=[ablation])
|
|
633
|
+
for fx in fixtures]
|
|
634
|
+
ablation_rows.append(_aggregate("practice_loop", (ablation,), per_fixture))
|
|
635
|
+
finally:
|
|
636
|
+
import shutil
|
|
637
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
638
|
+
|
|
639
|
+
budgets = {r["total_metered_budget"] for r in arm_rows} | {r["total_metered_budget"] for r in ablation_rows}
|
|
640
|
+
budget_match = all(r["budget_match"] for r in arm_rows + ablation_rows)
|
|
641
|
+
|
|
642
|
+
# ---- the key comparisons (synthesis §5 falsifiers) ------------------- #
|
|
643
|
+
practice = next(r for r in arm_rows if r["arm"] == "practice_loop" and not r["ablations"])
|
|
644
|
+
a3 = next(r for r in ablation_rows if r["ablations"] == ["a3_no_consolidation"])
|
|
645
|
+
a2 = next(r for r in ablation_rows if r["ablations"] == ["a2_no_spacing"])
|
|
646
|
+
comparison = _verdict(practice, a2, a3, arm_rows)
|
|
647
|
+
|
|
648
|
+
payload = {
|
|
649
|
+
"kind": AGENT_LEARNING_CAPSTONE_RESULT_KIND,
|
|
650
|
+
"experiment": {
|
|
651
|
+
"fixtures": fixture_names,
|
|
652
|
+
"equal_total_budget": total_budget,
|
|
653
|
+
"seed": seed,
|
|
654
|
+
"budget_match": budget_match,
|
|
655
|
+
"metered_budgets_observed": sorted(budgets),
|
|
656
|
+
"headline_metric": "retention_after_interference",
|
|
657
|
+
"arms": arm_rows,
|
|
658
|
+
"ablations": ablation_rows,
|
|
659
|
+
"key_comparison": comparison,
|
|
660
|
+
},
|
|
661
|
+
}
|
|
662
|
+
return public_payload(payload, kind=AGENT_LEARNING_CAPSTONE_RESULT_KIND)
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
def _aggregate(arm: str, ablations: Tuple[str, ...], per_fixture: Sequence[Mapping[str, Any]]) -> dict:
|
|
666
|
+
ret = [r["retention_after_interference"] for r in per_fixture]
|
|
667
|
+
bf = [r["best_found"] for r in per_fixture]
|
|
668
|
+
stab = [r["stability"] for r in per_fixture]
|
|
669
|
+
plas = [r["plasticity"] for r in per_fixture]
|
|
670
|
+
consumed = max(r["total_metered_budget"] for r in per_fixture)
|
|
671
|
+
detected = [r["detection_latency"].get("detected") for r in per_fixture]
|
|
672
|
+
return {
|
|
673
|
+
"arm": arm,
|
|
674
|
+
"ablations": list(ablations),
|
|
675
|
+
"mean_retention": round(statistics.fmean(ret), 6),
|
|
676
|
+
"mean_best_found": round(statistics.fmean(bf), 6),
|
|
677
|
+
"mean_stability": round(statistics.fmean(stab), 6),
|
|
678
|
+
"mean_plasticity": round(statistics.fmean(plas), 6),
|
|
679
|
+
"retention_by_fixture": {r["fixture"]: r["retention_after_interference"] for r in per_fixture},
|
|
680
|
+
"standing_detection_any": any(detected),
|
|
681
|
+
"total_metered_budget": consumed,
|
|
682
|
+
"budget_match": all(r["budget_match"] for r in per_fixture),
|
|
683
|
+
"per_fixture": list(per_fixture),
|
|
684
|
+
}
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def _verdict(practice: Mapping[str, Any], a2: Mapping[str, Any], a3: Mapping[str, Any],
|
|
688
|
+
arm_rows: Sequence[Mapping[str, Any]]) -> dict:
|
|
689
|
+
"""The pre-registered falsifier evaluation (synthesis §5)."""
|
|
690
|
+
p_ret = practice["mean_retention"]
|
|
691
|
+
a3_ret = a3["mean_retention"]
|
|
692
|
+
a2_ret = a2["mean_retention"]
|
|
693
|
+
lift_vs_a3 = round(p_ret - a3_ret, 6)
|
|
694
|
+
lift_vs_a2 = round(p_ret - a2_ret, 6)
|
|
695
|
+
# a meaningful lift: practice retains materially more than no-consolidation.
|
|
696
|
+
meaningful = lift_vs_a3 >= 0.05
|
|
697
|
+
if meaningful:
|
|
698
|
+
verdict = "LIFT_REAL"
|
|
699
|
+
note = ("spaced-regression-replay shows a retention lift vs no-consolidation "
|
|
700
|
+
"at equal budget; consolidation is load-bearing on these fixtures")
|
|
701
|
+
elif abs(lift_vs_a3) < 0.05 and abs(lift_vs_a2) < 0.05:
|
|
702
|
+
verdict = "NULL"
|
|
703
|
+
note = ("A3 retains equally — consolidation is decoration on these fixtures "
|
|
704
|
+
"(report the null per pre-registered falsifier)")
|
|
705
|
+
else:
|
|
706
|
+
verdict = "INCONCLUSIVE"
|
|
707
|
+
note = "lift present vs one ablation but not the other; inspect per-fixture rows"
|
|
708
|
+
return {
|
|
709
|
+
"verdict": verdict,
|
|
710
|
+
"note": note,
|
|
711
|
+
"practice_retention": p_ret,
|
|
712
|
+
"a3_no_consolidation_retention": a3_ret,
|
|
713
|
+
"a2_no_spacing_retention": a2_ret,
|
|
714
|
+
"retention_lift_vs_a3_no_consolidation": lift_vs_a3,
|
|
715
|
+
"retention_lift_vs_a2_no_spacing": lift_vs_a2,
|
|
716
|
+
"vs_search_arms": {
|
|
717
|
+
r["arm"]: r["mean_retention"] for r in arm_rows if r["arm"] != "practice_loop"
|
|
718
|
+
},
|
|
719
|
+
"supports_paper": verdict == "LIFT_REAL",
|
|
720
|
+
}
|