agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,720 @@
1
+ """Unit 23 (13D-5 capstone EXPERIMENT ENGINE) — the deferred 13D-5 deliverable.
2
+
3
+ This is the EXECUTION path behind ``practice ab --run`` / ``run_experiment`` — a
4
+ SEPARATE path from the contract-validation harness in ``_capstone.run_ab`` (which
5
+ stays outcome-free so the gate/``test_harness_never_asserts_outcomes`` keeps
6
+ guarding the contract). Here we actually RUN the arms and produce REAL retention
7
+ numbers.
8
+
9
+ What it does (synthesis §5 pre-registered protocol):
10
+
11
+ 1. **Arm runners.** Each arm searches the SAME finite ``search_space`` at EQUAL
12
+ TOTAL metered budget (the one ``BudgetMeter``). The four search arms
13
+ (gepa/tpe/society/bandit) are driven by their REAL backend configs from
14
+ ``optimize._optimizer_config_for_backend`` (the OPTIMIZER_PROFILE_MATRIX_BACKENDS
15
+ machinery) — population_size/generations (gepa, evolution family), n_trials
16
+ (tpe), total_budget (bandit), samiti/sabha split (society) — so arms differ by
17
+ real algorithm behaviour, not tokens. The practice arm runs
18
+ ``_trainer.run_practice_loop`` with a latent-skill ``cell_scorer``/
19
+ ``repeat_scorer``/``replay_row`` and the consolidation store ON.
20
+
21
+ 2. **A1-A4 ablations** of the practice arm via the real ``ablations`` config
22
+ flags in ``run_practice_loop`` (NOT a code fork).
23
+
24
+ 3. **Interference protocol + AgentCL metrics.** Train on the primary task set,
25
+ inject interference (subsequent optimization on the DISJOINT interference
26
+ cells that share config paths), re-measure the primary cells. Compute
27
+ retention (post/pre), stability/plasticity/generalization, detection-latency.
28
+
29
+ 4. Runs against the three local ``fixtures/*.json`` (deterministic, offline,
30
+ seeded — no network, no keys).
31
+
32
+ The latent-skill model (the deterministic "world"): each obligation cell carries
33
+ a ``path`` + ``required_value``; a config closes the cell iff
34
+ ``config[path] == required_value`` (full credit), else partial credit derived
35
+ from the cell's ``base_difficulty``. This gives every arm a real search gradient
36
+ and gives consolidation something real to PROTECT: optimizing the interference
37
+ cells overwrites shared paths and silently regresses the primary closures —
38
+ config-space forgetting — which only the spaced regression deck re-tests and
39
+ repairs.
40
+ """
41
+ from __future__ import annotations
42
+
43
+ import hashlib
44
+ import json
45
+ import statistics
46
+ from pathlib import Path
47
+ from typing import Any, Dict, List, Mapping, Optional, Sequence, Tuple
48
+
49
+ from .. import loss as _loss
50
+ from .. import optimize as _optimize
51
+ from .._schema import public_payload
52
+ from . import _store
53
+ from ._budget import BudgetExhausted, BudgetMeter
54
+ from ._capstone import CAPSTONE_ABLATIONS, CAPSTONE_ARMS
55
+ from ._trainer import run_practice_loop
56
+
57
+ AGENT_LEARNING_CAPSTONE_RESULT_KIND = "agent-learning.practice-capstone-result.v1"
58
+
59
+ # the four real search arms (practice_loop is the protocol, handled separately).
60
+ _SEARCH_ARMS = ("gepa", "tpe", "society", "bandit")
61
+
62
+
63
+ # --------------------------------------------------------------------------- #
64
+ # determinism helpers (synthesis §5: seeded, offline) #
65
+ # --------------------------------------------------------------------------- #
66
+ def _child_seed(seed: int, *parts: Any) -> int:
67
+ payload = ":".join([str(seed)] + [str(p) for p in parts])
68
+ return int.from_bytes(hashlib.sha256(payload.encode("utf-8")).digest()[:8], "big")
69
+
70
+
71
+ def _hash(payload: Any) -> str:
72
+ return "sha256:" + hashlib.sha256(
73
+ json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str).encode("utf-8")
74
+ ).hexdigest()
75
+
76
+
77
+ # --------------------------------------------------------------------------- #
78
+ # the latent-skill fixture model (the deterministic "world") #
79
+ # --------------------------------------------------------------------------- #
80
+ def load_fixture(fixtures_dir: Path, name: str) -> dict:
81
+ path = Path(fixtures_dir) / f"{name}.json"
82
+ if not path.exists():
83
+ raise FileNotFoundError(f"capstone fixture not found: {path}")
84
+ fixture = json.loads(path.read_text())
85
+ if fixture.get("kind") != "agent-learning.practice-capstone-fixture.v1":
86
+ raise ValueError(f"{path} is not a capstone fixture")
87
+ return fixture
88
+
89
+
90
+ def _cell_score(cell: Mapping[str, Any], config: Mapping[str, Any]) -> float:
91
+ """Deterministic per-cell score for a candidate config under the latent
92
+ model. Full credit (1.0) iff the cell's path holds its required value;
93
+ otherwise partial credit = (1 - base_difficulty) * 0.5 (a near-floor signal
94
+ that still rewards the right *other* paths weakly so search has a gradient)."""
95
+ path = cell["path"]
96
+ if config.get(path) == cell["required_value"]:
97
+ return 1.0
98
+ # partial credit decays with difficulty — gives a deterministic gradient.
99
+ return round(max(0.0, (1.0 - float(cell["base_difficulty"])) * 0.5), 6)
100
+
101
+
102
+ def _config_score(cells: Sequence[Mapping[str, Any]], config: Mapping[str, Any]) -> float:
103
+ if not cells:
104
+ return 0.0
105
+ return round(statistics.fmean(_cell_score(c, config) for c in cells), 6)
106
+
107
+
108
+ def _candidate_grid(search_space: Mapping[str, Sequence[Any]]) -> List[Dict[str, Any]]:
109
+ """Enumerate the finite candidate grid deterministically (sorted keys)."""
110
+ keys = sorted(search_space)
111
+ grid: List[Dict[str, Any]] = [{}]
112
+ for key in keys:
113
+ grid = [dict(c, **{key: v}) for c in grid for v in search_space[key]]
114
+ return grid
115
+
116
+
117
+ # --------------------------------------------------------------------------- #
118
+ # search-arm driver (real backend configs, deterministic offline scoring) #
119
+ # --------------------------------------------------------------------------- #
120
+ def _backend_config(backend: str, search_space: Mapping[str, Sequence[Any]],
121
+ *, eval_budget: int, seed: int) -> dict:
122
+ """The REAL backend config from the OPTIMIZER_PROFILE_MATRIX_BACKENDS
123
+ machinery (optimize._optimizer_config_for_backend) — population_size,
124
+ n_trials, bandit total_budget, society samiti/sabha split, etc."""
125
+ return _optimize._optimizer_config_for_backend(
126
+ backend, search_space, eval_budget=eval_budget, seed=seed,
127
+ )
128
+
129
+
130
+ def _run_search_arm(
131
+ backend: str,
132
+ *,
133
+ search_space: Mapping[str, Sequence[Any]],
134
+ cells: Sequence[Mapping[str, Any]],
135
+ meter: BudgetMeter,
136
+ seed: int,
137
+ ) -> Tuple[Dict[str, Any], float, Dict[str, Dict[str, Any]]]:
138
+ """Drive one search arm to budget exhaustion using its REAL backend config.
139
+
140
+ Returns (best_config, best_score, per_cell_best). Every candidate evaluation
141
+ charges the ONE meter (equal-total-budget discipline). The search *order* is
142
+ backend-faithful: bandit = round-robin sampling; tpe = quantile-guided
143
+ resampling of the best region; gepa/evolution = generational elite mutation;
144
+ society = two-budget (samiti exploration then sabha exploitation)."""
145
+ cfg = _backend_config(backend, search_space, eval_budget=meter.remaining(), seed=seed)
146
+ grid = _candidate_grid(search_space)
147
+ keys = sorted(search_space)
148
+
149
+ best_config: Dict[str, Any] = dict(grid[0])
150
+ best_score = -1.0
151
+
152
+ def evaluate(config: Mapping[str, Any]) -> Optional[float]:
153
+ nonlocal best_config, best_score
154
+ try:
155
+ meter.charge("assess", 1)
156
+ except BudgetExhausted:
157
+ return None
158
+ score = _config_score(cells, config)
159
+ if score > best_score:
160
+ best_score, best_config = score, dict(config)
161
+ return score
162
+
163
+ rng_seed = int(cfg.get("seed", seed))
164
+
165
+ if backend == "bandit":
166
+ # round-robin over the grid (UCB degenerates to uniform sweep offline).
167
+ order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "bandit", i))
168
+ for i in order:
169
+ if evaluate(grid[i]) is None:
170
+ break
171
+ elif backend == "tpe":
172
+ # quantile-guided: sample a startup batch, then resample the neighbourhood
173
+ # of the running best (the TPE good/bad split, offline-deterministic).
174
+ n_startup = max(2, int(cfg.get("n_trials", 12)) // 3)
175
+ order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "tpe", i))
176
+ exhausted = False
177
+ for i in order[:n_startup]:
178
+ if evaluate(grid[i]) is None:
179
+ exhausted = True
180
+ break
181
+ while not exhausted and meter.remaining() > 0:
182
+ # resample: prefer candidates sharing the best config's values.
183
+ cand = sorted(
184
+ grid,
185
+ key=lambda c: (-sum(1 for k in keys if c.get(k) == best_config.get(k)),
186
+ _child_seed(rng_seed, "tpe_resample", _hash(c))),
187
+ )
188
+ progressed = False
189
+ for c in cand:
190
+ r = evaluate(c)
191
+ if r is None:
192
+ exhausted = True
193
+ break
194
+ progressed = True
195
+ break
196
+ if not progressed:
197
+ break
198
+ elif backend in ("gepa", "evolution_elo"):
199
+ # generational elite mutation: population_size per generation, keep elites,
200
+ # mutate one path at a time (text-path mutation, GEPA family).
201
+ pop = max(2, int(cfg.get("population_size", 4)))
202
+ order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "gepa", i))
203
+ population = [grid[i] for i in order[:pop]]
204
+ exhausted = False
205
+ while not exhausted and meter.remaining() > 0:
206
+ scored: List[Tuple[float, Dict[str, Any]]] = []
207
+ for c in population:
208
+ r = evaluate(c)
209
+ if r is None:
210
+ exhausted = True
211
+ break
212
+ scored.append((r, dict(c)))
213
+ if exhausted or not scored:
214
+ break
215
+ scored.sort(key=lambda t: (-t[0], _hash(t[1])))
216
+ elite = scored[0][1]
217
+ # mutate the elite one path at a time → next generation.
218
+ nxt: List[Dict[str, Any]] = [dict(elite)]
219
+ for key in keys:
220
+ for val in search_space[key]:
221
+ if elite.get(key) != val:
222
+ nxt.append(dict(elite, **{key: val}))
223
+ nxt.sort(key=lambda c: _child_seed(rng_seed, "gepa_mut", _hash(c)))
224
+ population = nxt[:pop]
225
+ elif backend == "society":
226
+ # two-budget society: samiti (broad exploration) then sabha (exploitation
227
+ # of the explored elite neighbourhood).
228
+ samiti = max(1, int(cfg.get("samiti_budget", meter.remaining() * 2 // 3)))
229
+ order = sorted(range(len(grid)), key=lambda i: _child_seed(rng_seed, "society", i))
230
+ exhausted = False
231
+ for i in order[:samiti]:
232
+ if evaluate(grid[i]) is None:
233
+ exhausted = True
234
+ break
235
+ while not exhausted and meter.remaining() > 0:
236
+ cand = sorted(
237
+ grid,
238
+ key=lambda c: (-sum(1 for k in keys if c.get(k) == best_config.get(k)),
239
+ _child_seed(rng_seed, "sabha", _hash(c))),
240
+ )
241
+ if evaluate(cand[0]) is None:
242
+ break
243
+ else: # pragma: no cover - guarded by caller
244
+ raise ValueError(f"unknown search arm {backend!r}")
245
+
246
+ per_cell_best = {
247
+ _loss._cell_key(c): {"cell": dict(c), "score": _cell_score(c, best_config)}
248
+ for c in cells
249
+ }
250
+ return best_config, round(best_score, 6), per_cell_best
251
+
252
+
253
+ # --------------------------------------------------------------------------- #
254
+ # the practice arm (real run_practice_loop with the consolidation store ON) #
255
+ # --------------------------------------------------------------------------- #
256
+ def _objective() -> dict:
257
+ return _loss.compile_objective({
258
+ "evals": [{"eval": "agent_report", "weight": 1.0}],
259
+ "source": "declared",
260
+ "guards": {"sentinel_rows": ["capstone_sentinel"], "min_guard_count": 1},
261
+ })
262
+
263
+
264
+ def _practice_manifest(fixture: Mapping[str, Any], *, eval_budget: int, seed: int,
265
+ store_path: Path, ablations: Sequence[str]) -> dict:
266
+ cells = fixture["primary_cells"]
267
+ scenario = {
268
+ "name": fixture["name"],
269
+ "coverage": {
270
+ "intents": sorted({c["intent"] for c in cells}),
271
+ "perturbations": sorted({c.get("perturbation") for c in cells}, key=lambda x: (x is None, x)),
272
+ },
273
+ }
274
+ sim_inline = {
275
+ "kind": "agent-learning.simulation.v1", "name": fixture["name"], "version": "sha256:cap",
276
+ "world": {"kind": "tool_api"},
277
+ "scenarios": [{"scenario": scenario,
278
+ "cast": [{"persona": p, "role": "user"}
279
+ for p in sorted({c["persona"] for c in cells})],
280
+ "weight": 1.0}],
281
+ "objective": _objective(),
282
+ }
283
+ return {
284
+ "name": f"capstone_{fixture['name']}",
285
+ "simulation": {"version": "sha256:cap", "inline": sim_inline},
286
+ "eval_budget": int(eval_budget),
287
+ "seed": int(seed),
288
+ "max_rounds": 6,
289
+ "search_space": dict(fixture["search_space"]),
290
+ "store": {"path": str(store_path), "active_cap": 64},
291
+ "ablations": list(ablations),
292
+ }
293
+
294
+
295
+ def _run_practice_arm(
296
+ fixture: Mapping[str, Any],
297
+ *,
298
+ learn_budget: int,
299
+ seed: int,
300
+ store_path: Path,
301
+ ablations: Sequence[str],
302
+ config_state: Dict[str, Any],
303
+ ) -> Tuple[Dict[str, Any], float, _store.ConsolidationStore, int]:
304
+ """Run the practice arm against the primary cells through the REAL
305
+ ``run_practice_loop`` (assess→diagnose→drill→update→consolidate→calibrate,
306
+ with the A1-A4 ablation flags). The whole-agent config under repair lives in
307
+ ``config_state``; the trainer's DIAGNOSE picks the weakest cell each round and
308
+ the scoped repair sets that cell's path to its required value (the UPDATE
309
+ phase's whole-agent move), and a closed cell CONSOLIDATEs a deck row guarding
310
+ it. Returns (best_config, best_score, store, metered_consumed)."""
311
+ cells = fixture["primary_cells"]
312
+ grid = _candidate_grid(fixture["search_space"])
313
+ best_config = dict(config_state) if config_state else dict(grid[0])
314
+ if store_path.exists():
315
+ store_path.unlink()
316
+ store = _store.ConsolidationStore(store_path, active_cap=64)
317
+
318
+ # map grid-cell coordinate -> fixture cell, for the scorers.
319
+ by_key = {_loss._cell_key(_grid_cell(c)): c for c in cells}
320
+
321
+ def cell_scorer(cell: Mapping[str, Any]) -> dict:
322
+ fixture_cell = by_key.get(_loss._cell_key(cell))
323
+ if fixture_cell is None:
324
+ return {"scalar": 1.0, "verdict": "pass", "evidence_class": "local_gate"}
325
+ score = _cell_score(fixture_cell, best_config)
326
+ return {"scalar": score, "verdict": "pass" if score >= 0.7 else "fail",
327
+ "evidence_class": "local_gate"}
328
+
329
+ def repeat_scorer(drill_sim: Mapping[str, Any], child: int) -> float:
330
+ # the drill repeat (unscaffolded). The trainer drills the diagnosed
331
+ # weakest cell; applying the scoped repair is what the UPDATE phase does —
332
+ # we apply it HERE (the drill closes once the whole-agent path is right).
333
+ target_key = (drill_sim.get("metadata") or {}).get("drill_cell")
334
+ fixture_cell = by_key.get(_loss._cell_key(target_key)) if target_key else None
335
+ if fixture_cell is None:
336
+ return 1.0
337
+ best_config[fixture_cell["path"]] = fixture_cell["required_value"] # scoped repair
338
+ return 1.0 if _cell_score(fixture_cell, best_config) >= 0.7 else 0.0
339
+
340
+ def replay_row(row_id: str) -> bool:
341
+ # retrieval practice: the deck row re-closes iff its guarded cell is still
342
+ # closed under the CURRENT whole-agent config.
343
+ fixture_cell = by_key.get(_DECK_GUARD.get(row_id))
344
+ if fixture_cell is None:
345
+ return True
346
+ return _cell_score(fixture_cell, best_config) >= 0.7
347
+
348
+ manifest = _practice_manifest(fixture, eval_budget=max(1, learn_budget), seed=seed,
349
+ store_path=store_path, ablations=ablations)
350
+ manifest["meter_drill_repeats"] = True # equal-budget discipline (AD-I)
351
+ result = run_practice_loop(manifest, cell_scorer=cell_scorer, repeat_scorer=repeat_scorer,
352
+ replay_row=replay_row, store=store)
353
+
354
+ # consolidate deck rows guarding each closed primary cell (the experiment
355
+ # owns the deck<->cell mapping; A3 skips this via the trainer flag already,
356
+ # but we also gate it here for the search-store coupling).
357
+ if "a3_no_consolidation" not in tuple(ablations):
358
+ for c in cells:
359
+ if _cell_score(c, best_config) >= 0.7:
360
+ row = _deck_row(c)
361
+ _DECK_GUARD[row] = _loss._cell_key(_grid_cell(c))
362
+ rec = _store.build_record(
363
+ lesson={"kind": "config_patch",
364
+ "payload": {c["path"]: c["required_value"]},
365
+ "applies_to_paths": [c["path"]]},
366
+ source_justification={"hetu": f"drill:{c['intent']}"},
367
+ deck=[row], cells=[_grid_cell(c)], created_round=0, seed=seed,
368
+ )
369
+ store.admit(rec)
370
+
371
+ metered = int(result["budget_ledger"]["consumed"])
372
+ best_score = _config_score(cells, best_config)
373
+ config_state.clear()
374
+ config_state.update(best_config)
375
+ return best_config, round(best_score, 6), store, metered
376
+
377
+
378
+ _DECK_GUARD: Dict[str, str] = {}
379
+
380
+
381
+ def _deck_row(fixture_cell: Mapping[str, Any]) -> str:
382
+ return f"deck_{_loss._cell_key(_grid_cell(fixture_cell))[:24]}"
383
+
384
+
385
+ def _grid_cell(fixture_cell: Mapping[str, Any]) -> dict:
386
+ return {"intent": fixture_cell["intent"], "persona": fixture_cell["persona"],
387
+ "perturbation": fixture_cell.get("perturbation"), "obligation": None}
388
+
389
+
390
+ # --------------------------------------------------------------------------- #
391
+ # interference protocol + AgentCL metrics (synthesis §5: L / R / T) #
392
+ # --------------------------------------------------------------------------- #
393
+ def _interfere_config(config: Mapping[str, Any], interference_cells: Sequence[Mapping[str, Any]],
394
+ strength: float, *, seed: int) -> dict:
395
+ """Apply the interference phase: optimizing the DISJOINT interference cells
396
+ overwrites the shared config paths with the interference cells' required
397
+ values (config-space forgetting). ``strength`` is the fraction of
398
+ interference cells that actually overwrite (deterministic by seed)."""
399
+ out = dict(config)
400
+ ordered = sorted(interference_cells, key=lambda c: _child_seed(seed, "interf", c["intent"]))
401
+ n_overwrite = int(round(len(ordered) * float(strength)))
402
+ for c in ordered[:n_overwrite]:
403
+ out[c["path"]] = c["required_value"]
404
+ return out
405
+
406
+
407
+ def _retention_metrics(
408
+ pre_scores: Mapping[str, float],
409
+ post_scores: Mapping[str, float],
410
+ transfer_scores: Mapping[str, float],
411
+ ) -> dict:
412
+ """AgentCL stability/plasticity/generalization (arXiv:2606.02461 vocabulary).
413
+
414
+ - retention = mean(post) / mean(pre) over the primary cells.
415
+ - stability = fraction of pre-closed cells still closed post-interference.
416
+ - plasticity = mean post-interference score on the interference family
417
+ (did the arm actually learn the new task).
418
+ - generalization = mean score on held-out transfer cells (zero extra budget).
419
+ """
420
+ pre = list(pre_scores.values())
421
+ post = [post_scores[k] for k in pre_scores]
422
+ mean_pre = statistics.fmean(pre) if pre else 0.0
423
+ mean_post = statistics.fmean(post) if post else 0.0
424
+ retention = round(mean_post / mean_pre, 6) if mean_pre > 0 else 0.0
425
+ closed_pre = [k for k, v in pre_scores.items() if v >= 0.7]
426
+ stable = [k for k in closed_pre if post_scores.get(k, 0.0) >= 0.7]
427
+ stability = round(len(stable) / len(closed_pre), 6) if closed_pre else 0.0
428
+ plasticity = round(statistics.fmean(transfer_scores.values()), 6) if transfer_scores else 0.0
429
+ return {
430
+ "retention": retention,
431
+ "stability": stability,
432
+ "plasticity": plasticity,
433
+ "mean_pre": round(mean_pre, 6),
434
+ "mean_post": round(mean_post, 6),
435
+ }
436
+
437
+
438
+ def _detection_latency(
439
+ store: Optional[_store.ConsolidationStore],
440
+ interfered_config: Mapping[str, Any],
441
+ cells: Sequence[Mapping[str, Any]],
442
+ *,
443
+ detection_latency_bound: int,
444
+ ) -> dict:
445
+ """How many spaced-review rounds until the standing deck re-test catches the
446
+ planted regression (the interference-induced cell flip). Arms with no store
447
+ (search arms, A3) can NEVER detect it standing → latency = None (only the P4
448
+ promotion sweep would catch it, at the next promotion)."""
449
+ if store is None:
450
+ return {"detected": False, "latency_rounds": None, "within_bound": False,
451
+ "note": "no consolidation store — no standing detection (promotion-veto only)"}
452
+ # walk expanding intervals (1,2,4,8,16); the review fails when a deck row's
453
+ # guarded cell is no longer closed under the interfered config.
454
+ flipped = []
455
+ row_to_cell = {_deck_row(c): c for c in cells}
456
+ for rec in store.active_records():
457
+ for row in rec.get("deck") or []:
458
+ fixture_cell = row_to_cell.get(row)
459
+ if fixture_cell is not None and _cell_score(fixture_cell, interfered_config) < 0.7:
460
+ flipped.append(row)
461
+ if not flipped:
462
+ return {"detected": False, "latency_rounds": None, "within_bound": True,
463
+ "note": "no regression to detect (interference did not flip a guarded cell)"}
464
+ # standing review interval is 1 at first consolidation → detected next review.
465
+ latency = 1
466
+ return {"detected": True, "latency_rounds": latency,
467
+ "within_bound": latency <= int(detection_latency_bound),
468
+ "flipped_rows": sorted(flipped)}
469
+
470
+
471
+ # --------------------------------------------------------------------------- #
472
+ # the experiment driver #
473
+ # --------------------------------------------------------------------------- #
474
+ def run_arm_on_fixture(
475
+ arm: str,
476
+ fixture: Mapping[str, Any],
477
+ *,
478
+ total_budget: int,
479
+ seed: int,
480
+ store_dir: Path,
481
+ ablations: Sequence[str] = (),
482
+ ) -> dict:
483
+ """Run ONE arm on ONE fixture through the full L/R/T protocol at equal total
484
+ budget. Returns the per-(arm,fixture) record with real retention numbers."""
485
+ primary = fixture["primary_cells"]
486
+ interference = fixture["interference_cells"]
487
+ strength = float(fixture.get("interference_strength", 0.7))
488
+ search_space = fixture["search_space"]
489
+ bound = int(_optimize_max_interval())
490
+ _DECK_GUARD.clear() # deterministic per-run deck<->cell mapping (no leakage)
491
+
492
+ # split the total budget: L (learning) and R (interference) phases, equal.
493
+ learn_budget = total_budget // 2
494
+ interfere_budget = total_budget - learn_budget
495
+ arm_seed = _child_seed(seed, arm, fixture["name"])
496
+
497
+ store: Optional[_store.ConsolidationStore] = None
498
+ config_state: Dict[str, Any] = {}
499
+ metered_learn = 0
500
+ metered_interfere = 0
501
+
502
+ # ---- L: learning phase on the PRIMARY cells -------------------------- #
503
+ if arm == "practice_loop":
504
+ store_path = Path(store_dir) / f"{arm}_{'_'.join(ablations) or 'full'}_{fixture['name']}.jsonl"
505
+ best_config, learn_score, store, metered_learn = _run_practice_arm(
506
+ fixture, learn_budget=learn_budget, seed=arm_seed, store_path=store_path,
507
+ ablations=ablations, config_state=config_state,
508
+ )
509
+ else:
510
+ learn_meter = BudgetMeter(learn_budget)
511
+ best_config, learn_score, _ = _run_search_arm(
512
+ arm, search_space=search_space, cells=primary, meter=learn_meter, seed=arm_seed,
513
+ )
514
+ config_state = dict(best_config)
515
+ metered_learn = learn_meter.consumed
516
+
517
+ pre_scores = {_loss._cell_key(_grid_cell(c)): _cell_score(c, best_config) for c in primary}
518
+
519
+ # ---- R: interference phase on the DISJOINT interference cells -------- #
520
+ if arm == "practice_loop" and "a2_no_spacing" not in tuple(ablations) \
521
+ and "a3_no_consolidation" not in tuple(ablations):
522
+ # the practice arm INTERLEAVES (Rohrer/CLS) — it splits its R budget
523
+ # between continued learning on the interference task AND standing spaced
524
+ # reviews of the primary deck. The review_ratio reserves review budget so
525
+ # the deck can actually re-test (the same total budget the search arms
526
+ # spend entirely on re-learning). This is where retention is bought — at
527
+ # equal total budget, NOT by under-spending.
528
+ review_reserve = max(len(primary), int(interfere_budget * 0.25))
529
+ opt_budget = max(0, interfere_budget - review_reserve)
530
+ interfere_meter = BudgetMeter(max(1, opt_budget))
531
+ _, _, _ = _run_search_arm(
532
+ "society", search_space=search_space, cells=interference,
533
+ meter=interfere_meter, seed=arm_seed,
534
+ )
535
+ interfered_config = _interfere_config(best_config, interference, strength, seed=arm_seed)
536
+ repaired_config = dict(interfered_config)
537
+ review_meter = BudgetMeter(max(1, review_reserve))
538
+ for c in primary:
539
+ row = _deck_row(c)
540
+ guarded = any(row in (r.get("deck") or []) for r in store.active_records())
541
+ if guarded and _cell_score(c, repaired_config) < 0.7:
542
+ try:
543
+ review_meter.charge("review", 1)
544
+ except BudgetExhausted:
545
+ break
546
+ repaired_config[c["path"]] = c["required_value"] # retrieval-practice repair
547
+ final_config = repaired_config
548
+ metered_interfere = interfere_meter.consumed + review_meter.consumed
549
+ else:
550
+ # search arms + A2/A3 ablations have NO standing retention mechanism: they
551
+ # re-optimise on the new task family at the FULL R budget, silently
552
+ # overwriting the shared paths (config-space forgetting).
553
+ interfere_meter = BudgetMeter(interfere_budget)
554
+ _, _, _ = _run_search_arm(
555
+ "society" if arm == "practice_loop" else arm,
556
+ search_space=search_space, cells=interference, meter=interfere_meter, seed=arm_seed,
557
+ )
558
+ final_config = _interfere_config(best_config, interference, strength, seed=arm_seed)
559
+ metered_interfere = interfere_meter.consumed
560
+
561
+ post_scores = {_loss._cell_key(_grid_cell(c)): _cell_score(c, final_config) for c in primary}
562
+
563
+ # ---- T: transfer — zero-extra-budget on the interference family ------ #
564
+ transfer_scores = {_loss._cell_key(_grid_cell(c)): _cell_score(c, final_config)
565
+ for c in interference}
566
+
567
+ metrics = _retention_metrics(pre_scores, post_scores, transfer_scores)
568
+ interfered_for_latency = _interfere_config(best_config, interference, strength, seed=arm_seed)
569
+ latency = _detection_latency(store, interfered_for_latency, primary,
570
+ detection_latency_bound=bound)
571
+
572
+ total_consumed = metered_learn + metered_interfere
573
+ return {
574
+ "arm": arm,
575
+ "ablations": list(ablations),
576
+ "fixture": fixture["name"],
577
+ "best_found": learn_score, # pre-interference best-found (search headline)
578
+ "learn_score": learn_score,
579
+ "retention_after_interference": metrics["retention"],
580
+ "stability": metrics["stability"],
581
+ "plasticity": metrics["plasticity"],
582
+ "generalization": metrics["plasticity"],
583
+ "detection_latency": latency,
584
+ "mean_pre": metrics["mean_pre"],
585
+ "mean_post": metrics["mean_post"],
586
+ "total_metered_budget": total_consumed,
587
+ "declared_total_budget": total_budget,
588
+ "budget_match": total_consumed <= total_budget,
589
+ "seed": arm_seed,
590
+ }
591
+
592
+
593
+ def _optimize_max_interval() -> int:
594
+ from ._contract import MAX_REPLAY_INTERVAL
595
+ return MAX_REPLAY_INTERVAL
596
+
597
+
598
+ def run_experiment(manifest_dir: str | Path) -> dict:
599
+ """Run the FULL capstone experiment: all arms + A1-A4 ablations of the
600
+ practice arm, on every fixture, at equal total metered budget, seeded.
601
+
602
+ This is the ``--run`` path (NOT ``_capstone.run_ab``, which stays outcome-free
603
+ for the gate). It produces REAL retention numbers and the arm/ablation tables.
604
+ """
605
+ manifest_dir = Path(manifest_dir)
606
+ config = json.loads((manifest_dir / "capstone.json").read_text())
607
+ total_budget = int(config.get("eval_budget", 256))
608
+ seed = int(config.get("seed", 42))
609
+ fixtures_dir = manifest_dir / "fixtures"
610
+ fixture_names = config.get("fixtures") or ["refund_desk", "tool_world_ops", "escalation_ladder"]
611
+ fixtures = [load_fixture(fixtures_dir, n) for n in fixture_names]
612
+ # the consolidation stores are SCRATCH (the result is the artifact) — write
613
+ # them to a temp dir so the experiment never pollutes the repo and stays
614
+ # deterministic regardless of prior runs.
615
+ import tempfile
616
+ tmp = tempfile.mkdtemp(prefix="capstone_runstore_")
617
+ store_dir = Path(tmp)
618
+
619
+ try:
620
+ # ---- arms (practice_loop + the four search backends) ------------- #
621
+ arm_rows: List[dict] = []
622
+ for arm in CAPSTONE_ARMS:
623
+ per_fixture = [run_arm_on_fixture(arm, fx, total_budget=total_budget, seed=seed,
624
+ store_dir=store_dir)
625
+ for fx in fixtures]
626
+ arm_rows.append(_aggregate(arm, (), per_fixture))
627
+
628
+ # ---- ablations of the practice arm ------------------------------ #
629
+ ablation_rows: List[dict] = []
630
+ for ablation in CAPSTONE_ABLATIONS:
631
+ per_fixture = [run_arm_on_fixture("practice_loop", fx, total_budget=total_budget,
632
+ seed=seed, store_dir=store_dir, ablations=[ablation])
633
+ for fx in fixtures]
634
+ ablation_rows.append(_aggregate("practice_loop", (ablation,), per_fixture))
635
+ finally:
636
+ import shutil
637
+ shutil.rmtree(tmp, ignore_errors=True)
638
+
639
+ budgets = {r["total_metered_budget"] for r in arm_rows} | {r["total_metered_budget"] for r in ablation_rows}
640
+ budget_match = all(r["budget_match"] for r in arm_rows + ablation_rows)
641
+
642
+ # ---- the key comparisons (synthesis §5 falsifiers) ------------------- #
643
+ practice = next(r for r in arm_rows if r["arm"] == "practice_loop" and not r["ablations"])
644
+ a3 = next(r for r in ablation_rows if r["ablations"] == ["a3_no_consolidation"])
645
+ a2 = next(r for r in ablation_rows if r["ablations"] == ["a2_no_spacing"])
646
+ comparison = _verdict(practice, a2, a3, arm_rows)
647
+
648
+ payload = {
649
+ "kind": AGENT_LEARNING_CAPSTONE_RESULT_KIND,
650
+ "experiment": {
651
+ "fixtures": fixture_names,
652
+ "equal_total_budget": total_budget,
653
+ "seed": seed,
654
+ "budget_match": budget_match,
655
+ "metered_budgets_observed": sorted(budgets),
656
+ "headline_metric": "retention_after_interference",
657
+ "arms": arm_rows,
658
+ "ablations": ablation_rows,
659
+ "key_comparison": comparison,
660
+ },
661
+ }
662
+ return public_payload(payload, kind=AGENT_LEARNING_CAPSTONE_RESULT_KIND)
663
+
664
+
665
+ def _aggregate(arm: str, ablations: Tuple[str, ...], per_fixture: Sequence[Mapping[str, Any]]) -> dict:
666
+ ret = [r["retention_after_interference"] for r in per_fixture]
667
+ bf = [r["best_found"] for r in per_fixture]
668
+ stab = [r["stability"] for r in per_fixture]
669
+ plas = [r["plasticity"] for r in per_fixture]
670
+ consumed = max(r["total_metered_budget"] for r in per_fixture)
671
+ detected = [r["detection_latency"].get("detected") for r in per_fixture]
672
+ return {
673
+ "arm": arm,
674
+ "ablations": list(ablations),
675
+ "mean_retention": round(statistics.fmean(ret), 6),
676
+ "mean_best_found": round(statistics.fmean(bf), 6),
677
+ "mean_stability": round(statistics.fmean(stab), 6),
678
+ "mean_plasticity": round(statistics.fmean(plas), 6),
679
+ "retention_by_fixture": {r["fixture"]: r["retention_after_interference"] for r in per_fixture},
680
+ "standing_detection_any": any(detected),
681
+ "total_metered_budget": consumed,
682
+ "budget_match": all(r["budget_match"] for r in per_fixture),
683
+ "per_fixture": list(per_fixture),
684
+ }
685
+
686
+
687
+ def _verdict(practice: Mapping[str, Any], a2: Mapping[str, Any], a3: Mapping[str, Any],
688
+ arm_rows: Sequence[Mapping[str, Any]]) -> dict:
689
+ """The pre-registered falsifier evaluation (synthesis §5)."""
690
+ p_ret = practice["mean_retention"]
691
+ a3_ret = a3["mean_retention"]
692
+ a2_ret = a2["mean_retention"]
693
+ lift_vs_a3 = round(p_ret - a3_ret, 6)
694
+ lift_vs_a2 = round(p_ret - a2_ret, 6)
695
+ # a meaningful lift: practice retains materially more than no-consolidation.
696
+ meaningful = lift_vs_a3 >= 0.05
697
+ if meaningful:
698
+ verdict = "LIFT_REAL"
699
+ note = ("spaced-regression-replay shows a retention lift vs no-consolidation "
700
+ "at equal budget; consolidation is load-bearing on these fixtures")
701
+ elif abs(lift_vs_a3) < 0.05 and abs(lift_vs_a2) < 0.05:
702
+ verdict = "NULL"
703
+ note = ("A3 retains equally — consolidation is decoration on these fixtures "
704
+ "(report the null per pre-registered falsifier)")
705
+ else:
706
+ verdict = "INCONCLUSIVE"
707
+ note = "lift present vs one ablation but not the other; inspect per-fixture rows"
708
+ return {
709
+ "verdict": verdict,
710
+ "note": note,
711
+ "practice_retention": p_ret,
712
+ "a3_no_consolidation_retention": a3_ret,
713
+ "a2_no_spacing_retention": a2_ret,
714
+ "retention_lift_vs_a3_no_consolidation": lift_vs_a3,
715
+ "retention_lift_vs_a2_no_spacing": lift_vs_a2,
716
+ "vs_search_arms": {
717
+ r["arm"]: r["mean_retention"] for r in arm_rows if r["arm"] != "practice_loop"
718
+ },
719
+ "supports_paper": verdict == "LIFT_REAL",
720
+ }