agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,102 @@
1
+ """Unit 9 (BBG U9 / ARCH §2d) — the spaced-replay schedule state machine.
2
+
3
+ The T1-T7 transition table as a PURE function. Public surface: ``due_reviews``
4
+ ONLY. This module has NO code path into promotion (the update module is the only
5
+ promotion invoker, obtaining rows exclusively from the store's deck union) — the
6
+ 13D-D7 boundary is structural, not disciplinary. There is deliberately NO import
7
+ of the promotion invoker here.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from typing import Any, List, Mapping
12
+
13
+ from ._contract import MAX_REPLAY_INTERVAL
14
+ from ._store import demote_ladder
15
+
16
+ _LADDER_ORDER = ("episodic", "instruction", "skill")
17
+
18
+
19
+ def _append_history(record: dict, round_no: int, event: str, outcome: str) -> None:
20
+ history = list(record.get("history") or [])
21
+ history.append({"round": int(round_no), "event": str(event), "outcome": str(outcome)})
22
+ record["history"] = history
23
+
24
+
25
+ def transition(record: Mapping[str, Any], event: str, round_no: int) -> dict:
26
+ """The T1-T7 table. ``event`` ∈ {'review_pass', 'review_fail', 'obsolete'}.
27
+ Returns a NEW record snapshot (every transition appends to history)."""
28
+ out = dict(record)
29
+ out["schedule"] = dict(out.get("schedule") or {})
30
+ schedule = out["schedule"]
31
+ ladder = out.get("ladder_state", "episodic")
32
+
33
+ if schedule.get("status") == "retired":
34
+ # T7: retired is terminal.
35
+ _append_history(out, round_no, event, "terminal_retired")
36
+ return out
37
+
38
+ if event == "obsolete":
39
+ # T5: obsolescence ⇒ retired "obsolete".
40
+ schedule["status"] = "retired"
41
+ schedule["retired_reason"] = "obsolete"
42
+ _append_history(out, round_no, "obsolete", "retired")
43
+ return out
44
+
45
+ if event == "review_pass":
46
+ # T1: interval ← min(2·i, 16); due ← round+interval; failures ← 0.
47
+ interval = int(schedule.get("interval_rounds", 1))
48
+ new_interval = min(2 * interval, MAX_REPLAY_INTERVAL)
49
+ schedule["interval_rounds"] = new_interval
50
+ schedule["due_round"] = int(round_no) + new_interval
51
+ schedule["consecutive_failures"] = 0
52
+ _append_history(out, round_no, "review_pass", f"interval->{new_interval}")
53
+ return out
54
+
55
+ if event == "review_fail":
56
+ failures = int(schedule.get("consecutive_failures", 0)) + 1
57
+ schedule["consecutive_failures"] = failures
58
+ if failures >= 2:
59
+ # T4: failures ≥ 2 ⇒ retired "repeated_failure".
60
+ schedule["status"] = "retired"
61
+ schedule["retired_reason"] = "repeated_failure"
62
+ _append_history(out, round_no, "review_fail", "retired_repeated_failure")
63
+ return out
64
+ if ladder == "episodic":
65
+ # T3: fail at episodic ⇒ retired "repeated_failure" (no rung below).
66
+ schedule["status"] = "retired"
67
+ schedule["retired_reason"] = "repeated_failure"
68
+ _append_history(out, round_no, "review_fail", "retired_episodic_fail")
69
+ return out
70
+ # T2: fail above episodic ⇒ demote one rung; interval←1; due←round+1.
71
+ demoted = demote_ladder(out)
72
+ out["ladder_state"] = demoted["ladder_state"]
73
+ schedule["interval_rounds"] = 1
74
+ schedule["due_round"] = int(round_no) + 1
75
+ _append_history(out, round_no, "review_fail", f"demote->{out['ladder_state']}")
76
+ return out
77
+
78
+ raise ValueError(f"unknown schedule event {event!r}")
79
+
80
+
81
+ def due_reviews(records: List[Mapping[str, Any]], round_no: int) -> List[dict]:
82
+ """Select due active records, sorted by (due_round, record_id) — deterministic
83
+ tie-break by content id. The ONLY public path from this module."""
84
+ due = [
85
+ dict(r)
86
+ for r in records
87
+ if r.get("schedule", {}).get("status") == "active"
88
+ and int(r.get("schedule", {}).get("due_round", 0)) <= int(round_no)
89
+ ]
90
+ due.sort(key=lambda r: (int(r["schedule"]["due_round"]), str(r.get("record_id", ""))))
91
+ return due
92
+
93
+
94
+ def mark_obsolete_if_path_left_space(
95
+ record: Mapping[str, Any], search_paths: List[str], round_no: int
96
+ ) -> dict:
97
+ """T5 obsolescence trigger helper: a deck row's config path leaves the search
98
+ space ⇒ retire 'obsolete'. (Caller supplies the current search space.)"""
99
+ applies = record.get("lesson", {}).get("applies_to_paths") or []
100
+ if applies and not any(p in set(search_paths) for p in applies):
101
+ return transition(record, "obsolete", round_no)
102
+ return dict(record)
@@ -0,0 +1,194 @@
1
+ """Unit 9 (BBG U9 / ARCH §2d) — the consolidation store.
2
+
3
+ Consolidated records (decks ARE frozen rows, AD-D): record ids use the
4
+ optimizer-space frozen-row idiom (``lesson_`` + 16-hex sorted-JSON digest).
5
+ Append-only JSONL (AD-G); state transitions are appended full-record snapshots
6
+ (latest-wins on read). ``full_deck()`` is the ONLY promotion-row source (13D-D7)
7
+ — the FULL union of all P4 frozen rows + every active record's complete deck,
8
+ regardless of schedule state. Cap admission REFUSES at active_cap (AD-H), never
9
+ evicts.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import hashlib
14
+ import json
15
+ from pathlib import Path
16
+ from typing import Any, Dict, List, Mapping, Optional
17
+
18
+ from ._contract import (
19
+ LADDER_STATES,
20
+ LESSON_ID_PREFIX,
21
+ LESSON_KINDS,
22
+ PRACTICE_REPLAY_INTERVALS,
23
+ PRACTICE_STORE_ACTIVE_CAP,
24
+ practice_store_path,
25
+ )
26
+
27
+ # fields that do NOT enter the record id (envelope/mutable state).
28
+ _NON_ID_FIELDS = {"record_id", "schedule", "history"}
29
+
30
+
31
+ def _sorted_json_digest(payload: Any) -> str:
32
+ return hashlib.sha256(
33
+ json.dumps(payload, sort_keys=True, default=str).encode("utf-8")
34
+ ).hexdigest()
35
+
36
+
37
+ def record_id(body: Mapping[str, Any]) -> str:
38
+ """The frozen-row idiom: lesson_ + 16-hex sorted-JSON digest over all
39
+ non-envelope fields (recipe-agreement with _expected_frozen_row_id)."""
40
+ payload = {k: v for k, v in body.items() if k not in _NON_ID_FIELDS}
41
+ return LESSON_ID_PREFIX + _sorted_json_digest(payload)[:16]
42
+
43
+
44
+ def build_record(
45
+ *,
46
+ lesson: Mapping[str, Any],
47
+ source_justification: Mapping[str, Any],
48
+ deck: List[str],
49
+ cells: List[Any],
50
+ created_round: int,
51
+ seed: int,
52
+ interval_rounds: int = 1,
53
+ due_round: Optional[int] = None,
54
+ provenance: Optional[Mapping[str, Any]] = None,
55
+ ladder_state: str = "episodic",
56
+ ) -> dict:
57
+ """Construct a consolidated record (ARCH §2d schema verbatim)."""
58
+ if ladder_state not in LADDER_STATES:
59
+ raise ValueError(f"ladder_state {ladder_state!r} not in {LADDER_STATES}")
60
+ if lesson.get("kind") not in LESSON_KINDS:
61
+ raise ValueError(f"lesson.kind {lesson.get('kind')!r} not in {LESSON_KINDS}")
62
+ if interval_rounds not in PRACTICE_REPLAY_INTERVALS:
63
+ raise ValueError(f"interval_rounds {interval_rounds!r} not in {PRACTICE_REPLAY_INTERVALS}")
64
+ body = {
65
+ "kind": "agent-learning.consolidated-lesson.v1",
66
+ "ladder_state": ladder_state,
67
+ "lesson": {
68
+ "kind": lesson["kind"],
69
+ "payload": lesson.get("payload"),
70
+ "applies_to_paths": list(lesson.get("applies_to_paths") or []),
71
+ },
72
+ "source_justification": dict(source_justification),
73
+ "deck": sorted(set(deck)),
74
+ "schedule": {
75
+ "interval_rounds": int(interval_rounds),
76
+ "due_round": int(due_round if due_round is not None else created_round + interval_rounds),
77
+ "consecutive_failures": 0,
78
+ "status": "active",
79
+ "retired_reason": None,
80
+ },
81
+ "cells": list(cells),
82
+ "history": [],
83
+ "created_round": int(created_round),
84
+ "seed": int(seed),
85
+ "provenance": dict(provenance or {}),
86
+ }
87
+ body["record_id"] = record_id(body)
88
+ return body
89
+
90
+
91
+ class ConsolidationStore:
92
+ """Append-only JSONL store under the user-owned home (AD-G)."""
93
+
94
+ def __init__(self, path: str | Path | None = None, *, active_cap: int = PRACTICE_STORE_ACTIVE_CAP) -> None:
95
+ self.path = practice_store_path(path)
96
+ self.active_cap = int(active_cap)
97
+
98
+ # --- IO (tolerant reader, AD-G) ----------------------------------------
99
+ def _read_snapshots(self) -> List[dict]:
100
+ if not self.path.exists():
101
+ return []
102
+ out: List[dict] = []
103
+ for line in self.path.read_text().splitlines():
104
+ line = line.strip()
105
+ if not line:
106
+ continue
107
+ try:
108
+ out.append(json.loads(line))
109
+ except json.JSONDecodeError:
110
+ continue # tolerant reader (live/_transcript.py philosophy)
111
+ return out
112
+
113
+ def _append(self, record: Mapping[str, Any]) -> None:
114
+ self.path.parent.mkdir(parents=True, exist_ok=True)
115
+ with self.path.open("a") as handle:
116
+ handle.write(json.dumps(record, sort_keys=True, default=str) + "\n")
117
+
118
+ def latest(self) -> Dict[str, dict]:
119
+ """Latest-wins by record_id over the append-only snapshots."""
120
+ out: Dict[str, dict] = {}
121
+ for snapshot in self._read_snapshots():
122
+ rid = snapshot.get("record_id")
123
+ if rid:
124
+ out[rid] = snapshot
125
+ return out
126
+
127
+ def active_records(self) -> List[dict]:
128
+ return [r for r in self.latest().values() if r.get("schedule", {}).get("status") == "active"]
129
+
130
+ # --- admission (cap refusal, AD-H) -------------------------------------
131
+ def admit(self, record: Mapping[str, Any]) -> dict:
132
+ """Admit a record, or refuse with cap_deferred at active_cap (refusal
133
+ over eviction — T6/AD-H)."""
134
+ rid = record.get("record_id") or record_id(record)
135
+ existing = self.latest()
136
+ if rid in existing:
137
+ self._append(dict(record))
138
+ return {"admitted": True, "record_id": rid, "reason": "updated"}
139
+ active_count = len([r for r in existing.values()
140
+ if r.get("schedule", {}).get("status") == "active"])
141
+ if active_count >= self.active_cap:
142
+ return {
143
+ "admitted": False,
144
+ "record_id": rid,
145
+ "status": "cap_deferred",
146
+ "reason": (
147
+ f"consolidation store at active_cap={self.active_cap}; admission "
148
+ "REFUSED (a slot frees only via retirement — refusal over eviction)"
149
+ ),
150
+ }
151
+ self._append(dict(record))
152
+ return {"admitted": True, "record_id": rid, "reason": "admitted"}
153
+
154
+ def update_record(self, record: Mapping[str, Any]) -> None:
155
+ """Append a full-record snapshot (state transition, never a rewrite)."""
156
+ self._append(dict(record))
157
+
158
+ # --- the ONLY promotion-row source (13D-D7) ----------------------------
159
+ def full_deck(self, *, frozen_rows: Optional[List[str]] = None) -> List[str]:
160
+ """The UNION of all P4 frozen rows + every active record's complete deck,
161
+ REGARDLESS of any record's schedule state. This is the 13D-D7 boundary."""
162
+ rows: set[str] = set(frozen_rows or [])
163
+ for record in self.active_records():
164
+ rows.update(record.get("deck") or [])
165
+ return sorted(rows)
166
+
167
+
168
+ def ladder_state(record: Mapping[str, Any]) -> str:
169
+ """Read the stored ladder state (never recomputed from history, AD-H)."""
170
+ return str(record.get("ladder_state") or "episodic")
171
+
172
+
173
+ # --- ladder transitions (up; ARCH §2d promotion ladder) --------------------
174
+ _LADDER_ORDER = ("episodic", "instruction", "skill")
175
+
176
+
177
+ def promote_ladder(record: Mapping[str, Any]) -> dict:
178
+ """Move one rung up (episodic→instruction→skill)."""
179
+ out = dict(record)
180
+ current = out.get("ladder_state", "episodic")
181
+ idx = _LADDER_ORDER.index(current)
182
+ if idx < len(_LADDER_ORDER) - 1:
183
+ out["ladder_state"] = _LADDER_ORDER[idx + 1]
184
+ return out
185
+
186
+
187
+ def demote_ladder(record: Mapping[str, Any]) -> dict:
188
+ """Move one rung down (skill→instruction→episodic)."""
189
+ out = dict(record)
190
+ current = out.get("ladder_state", "episodic")
191
+ idx = _LADDER_ORDER.index(current)
192
+ if idx > 0:
193
+ out["ladder_state"] = _LADDER_ORDER[idx - 1]
194
+ return out
@@ -0,0 +1,245 @@
1
+ """Unit 13 (BBG U13 / ARCH §2d phases 5-6) — the six-phase practice driver.
2
+
3
+ ``run_practice_loop(manifest) -> dict``: the outer loop per round —
4
+ rank deficits → drill (interleaved with due reviews at review_ratio, reviews
5
+ ONLY between promotions) → update → consolidate → re-assess; bounded by
6
+ max_rounds and the meter. Entry-point checks (before ANY budget): derived-
7
+ objective refusal (AD-F), eval_budget presence, extension admission. Determinism
8
+ per ARCH §2d (all child seeds derived; every artifact carries seed + parent
9
+ hashes). Emits ``agent-learning.practice-result.v1`` through public_payload, so
10
+ every episode lands a telemetry ledger row with zero new telemetry code.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import json
16
+ from typing import Any, Callable, List, Mapping, Optional
17
+
18
+ from .._schema import public_payload
19
+ from .. import loss as _loss
20
+ from . import _assess, _diagnose, _drill, _schedule, _store, _update
21
+ from ._budget import BudgetExhausted, BudgetMeter
22
+ from ._contract import (
23
+ AGENT_LEARNING_PRACTICE_RESULT_KIND,
24
+ BUDGET_PLAN,
25
+ DEFAULT_INNER_OPERATOR_BACKEND,
26
+ DEFAULT_MAX_ROUNDS,
27
+ PRACTICE_ABLATIONS,
28
+ REVIEW_RATIO,
29
+ SCAFFOLD_FADE_DEFAULT,
30
+ ZPD_BAND,
31
+ )
32
+
33
+
34
+ def _hash(payload: Mapping[str, Any]) -> str:
35
+ return "sha256:" + hashlib.sha256(
36
+ json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str).encode("utf-8")
37
+ ).hexdigest()
38
+
39
+
40
+ class PracticeRefusal(ValueError):
41
+ """Raised at the entry point before any budget is spent."""
42
+
43
+
44
+ def run_practice_loop(
45
+ manifest: Mapping[str, Any],
46
+ *,
47
+ cell_scorer: Optional[Callable[[Mapping[str, Any]], Mapping[str, Any]]] = None,
48
+ repeat_scorer: Optional[Callable[[Mapping[str, Any], int], float]] = None,
49
+ replay_row: Optional[Callable[[str], bool]] = None,
50
+ store: Optional[_store.ConsolidationStore] = None,
51
+ extension_admission_check: Optional[Callable[[], Optional[dict]]] = None,
52
+ ) -> dict:
53
+ """Run the practice loop deterministically. Scorers are injected so the gate
54
+ and capstone run offline; production wires them to the simulation engine."""
55
+ practice = dict(manifest.get("practice") or manifest)
56
+ sim_block = dict(practice.get("simulation") or {})
57
+ simulation = dict(sim_block.get("inline") or {})
58
+ objective = simulation.get("objective")
59
+
60
+ # --- entry-point refusals (BEFORE any budget) --------------------------
61
+ # 1. derived-objective refusal (AD-F).
62
+ _loss.refuse_derived_for_training(objective)
63
+ # 2. eval_budget presence (else budget_undeclared).
64
+ eval_budget = practice.get("eval_budget")
65
+ if not isinstance(eval_budget, int) or isinstance(eval_budget, bool) or eval_budget < 1:
66
+ raise PracticeRefusal("budget_undeclared: eval_budget is required and must be an int >= 1")
67
+ # 3. extension admission for the inner operator (before any phase spends).
68
+ if extension_admission_check is not None:
69
+ refusal = extension_admission_check()
70
+ if refusal is not None and not refusal.get("admitted", True):
71
+ raise PracticeRefusal(
72
+ f"extension_evidence_inadmissible: {refusal.get('reason')}"
73
+ )
74
+
75
+ seed = practice.get("seed")
76
+ if seed is None:
77
+ raise PracticeRefusal("seed is required (declared-MANDATORY pair: eval_budget + seed)")
78
+ seed = int(seed)
79
+
80
+ max_rounds = int(practice.get("max_rounds", DEFAULT_MAX_ROUNDS))
81
+ budget_plan = tuple(practice.get("budget_plan") or BUDGET_PLAN)
82
+ review_ratio = float(practice.get("review_ratio", REVIEW_RATIO))
83
+ zpd = dict(practice.get("zpd") or {})
84
+ band = tuple(zpd.get("band") or ZPD_BAND)
85
+ k = int(zpd.get("k", 8))
86
+ icc_floor = float(zpd.get("icc_floor", 0.5))
87
+ scaffold_fade = dict(practice.get("scaffold_fade") or {})
88
+ fade = tuple(scaffold_fade.get("intensities") or SCAFFOLD_FADE_DEFAULT)
89
+ inner_operator = dict(practice.get("inner_operator") or {})
90
+ operator_backend = str(inner_operator.get("backend", DEFAULT_INNER_OPERATOR_BACKEND))
91
+ frozen_rows = list(practice.get("frozen_rows") or [])
92
+ search_space = dict(practice.get("search_space") or manifest.get("search_space") or {})
93
+
94
+ # --- ablation knobs (13D-5 capstone; additive, default = full loop) -----
95
+ # Real config flags that change behaviour, never labels. Unknown tokens are
96
+ # a contract error (the experiment must not silently no-op an ablation).
97
+ ablations = tuple(practice.get("ablations") or ())
98
+ for ablation in ablations:
99
+ if ablation not in PRACTICE_ABLATIONS:
100
+ raise PracticeRefusal(
101
+ f"unknown ablation {ablation!r}; must be one of {PRACTICE_ABLATIONS}"
102
+ )
103
+ a1_no_zpd = "a1_no_zpd" in ablations
104
+ a2_no_spacing = "a2_no_spacing" in ablations
105
+ a3_no_consolidation = "a3_no_consolidation" in ablations
106
+ a4_no_calibration = "a4_no_calibration" in ablations
107
+ # A4 needs the learned-gate; the loop reports per-cell stop signals so an
108
+ # external driver (the experiment engine) can stop a learned cell early.
109
+ learned_cells: set = set()
110
+ # Equal-total-budget discipline (synthesis §5 / AD-I): every ZPD repeat is a
111
+ # scored evaluation and MUST charge the meter. Opt-in (default off) so the
112
+ # gate/determinism-fixture path stays byte-identical; the capstone experiment
113
+ # turns it on so the practice arm meters the same currency as the search arms.
114
+ meter_drill_repeats = bool(practice.get("meter_drill_repeats", False))
115
+
116
+ meter = BudgetMeter(eval_budget, budget_plan=budget_plan)
117
+ if store is None:
118
+ store_knob = dict(practice.get("store") or {})
119
+ store = _store.ConsolidationStore(store_knob.get("path"),
120
+ active_cap=int(store_knob.get("active_cap", 64)))
121
+
122
+ # default deterministic scorers (all-pass) if not injected.
123
+ cell_scorer = cell_scorer or (lambda cell: {"scalar": 1.0, "verdict": "pass", "evidence_class": "local_gate"})
124
+ repeat_scorer = repeat_scorer or (lambda sim, s: 1.0)
125
+ replay_row = replay_row or (lambda row: True)
126
+
127
+ rounds: List[dict] = []
128
+ parent_report_hash: Optional[str] = None
129
+ stop_reason = "max_rounds"
130
+
131
+ try:
132
+ for round_no in range(max_rounds):
133
+ # ASSESS
134
+ report = _assess.assess(
135
+ simulation, objective, meter=meter, round_no=round_no, seed=seed,
136
+ cell_scorer=cell_scorer, parent_report_hash=parent_report_hash,
137
+ )
138
+ parent_report_hash = _hash(report)
139
+ # DIAGNOSE
140
+ deficits = _diagnose.diagnose(report, search_space=search_space)
141
+ # DRILL (interleaved with due reviews between promotions).
142
+ # A2 no-spacing: NO standing between-promotion reviews (the deck
143
+ # only ever replays at the promotion sweep — replay-only-at-promotion).
144
+ if not a2_no_spacing:
145
+ due = _schedule.due_reviews(store.active_records(), round_no)
146
+ review_slots = int(len(deficits["deficits"]) * review_ratio)
147
+ for review_rec in due[:review_slots]:
148
+ meter.charge("review", 1)
149
+ review_pass = all(replay_row(r) for r in review_rec.get("deck") or [])
150
+ event = "review_pass" if review_pass else "review_fail"
151
+ store.update_record(_schedule.transition(review_rec, event, round_no))
152
+
153
+ drill_records: List[dict] = []
154
+ update_records: List[dict] = []
155
+ for deficit in deficits["deficits"]:
156
+ # A4 no-calibration: fixed-k, never stop a learned cell early —
157
+ # so learned cells are NOT pruned and keep consuming drill budget.
158
+ if not a4_no_calibration and _loss._cell_key(deficit.get("cell") or {}) in learned_cells:
159
+ continue
160
+ # charge the ZPD repeats to the meter (opt-in, AD-I) BEFORE the
161
+ # drill runs — k repeats per drill are k scored evaluations.
162
+ if meter_drill_repeats:
163
+ meter.charge("drill", k)
164
+ drill = _drill.drill(
165
+ deficit, _drill_simulation(simulation, deficit), seed=seed, round_no=round_no,
166
+ repeat_scorer=repeat_scorer, fade_intensities=fade, k=k,
167
+ icc_floor=icc_floor, band=band,
168
+ )
169
+ drill_records.append(drill)
170
+ # A1 no-ZPD: drill is NOT ZPD-filtered — an unstable/out-of-band
171
+ # drill is still promoted to UPDATE (the full loop quarantines it).
172
+ if not a1_no_zpd and drill["zpd_measurement"]["verdict"] == "unstable":
173
+ continue
174
+ # CALIBRATE (A4 disables): mark a cell learned when its
175
+ # unscaffolded pass-rate clears the band ceiling at stable ICC.
176
+ if not a4_no_calibration:
177
+ zpd = drill["zpd_measurement"]
178
+ if (zpd["unscaffolded_pass_rate"] >= float(band[1])
179
+ and zpd["icc"] >= icc_floor):
180
+ learned_cells.add(_loss._cell_key(deficit.get("cell") or {}))
181
+ # UPDATE (the D7 promotion sweep)
182
+ upd = _update.update(
183
+ deficit, allowed_layer=deficit.get("harness_layer", "execution"),
184
+ allowed_paths=deficit.get("search_paths") or [],
185
+ proposals=[{"patch": {}, "justification": {"hetu": "drill"}}],
186
+ store=store, frozen_rows=frozen_rows, replay_row=replay_row,
187
+ meter=meter, operator_backend=operator_backend,
188
+ )
189
+ update_records.append(upd)
190
+ # CONSOLIDATE (admit a lesson; cap ⇒ cap_deferred).
191
+ # A3 no-consolidation: skip the consolidate phase entirely —
192
+ # lessons stay episodic-in-the-run and are NEVER admitted to the
193
+ # store, so there is no spaced deck to protect against drift.
194
+ if not a3_no_consolidation and upd["promotion_sweep"]["all_closed"]:
195
+ rec = _store.build_record(
196
+ lesson={"kind": "config_patch", "payload": {}, "applies_to_paths": deficit.get("search_paths") or []},
197
+ source_justification=upd.get("selected_candidate", {}).get("justification", {}) if upd.get("selected_candidate") else {},
198
+ deck=list(frozen_rows), cells=[deficit.get("cell")],
199
+ created_round=round_no, seed=seed,
200
+ )
201
+ store.admit(rec)
202
+ rounds.append({
203
+ "round": round_no,
204
+ "report": report,
205
+ "deficits": deficits,
206
+ "drills": drill_records,
207
+ "updates": update_records,
208
+ })
209
+ except BudgetExhausted:
210
+ stop_reason = "budget_exhausted"
211
+
212
+ result = {
213
+ "kind": AGENT_LEARNING_PRACTICE_RESULT_KIND,
214
+ "name": practice.get("name") or manifest.get("name"),
215
+ "seed": seed,
216
+ "simulation_version": simulation.get("version"),
217
+ "objective_version": (objective or {}).get("version"),
218
+ "stop_reason": stop_reason,
219
+ "rounds_completed": len(rounds),
220
+ "ablations": list(ablations),
221
+ "learned_cell_count": len(learned_cells),
222
+ "budget_ledger": meter.ledger(),
223
+ "rounds": rounds,
224
+ # headline — AgentCL stability/plasticity/generalization, never best-found.
225
+ "retention_and_transfer_at_equal_budget": {
226
+ "stability": None, "plasticity": None, "generalization": None,
227
+ },
228
+ "promotion_veto_boundary": (
229
+ "all frozen rows replay at every promotion regardless of schedule state (13D-D7)"
230
+ ),
231
+ "detection_latency": {"measured": None, "declared_bound": practice.get("schedule", {}).get("detection_latency_bound")},
232
+ }
233
+ return public_payload(result, kind=AGENT_LEARNING_PRACTICE_RESULT_KIND)
234
+
235
+
236
+ def _drill_simulation(simulation: Mapping[str, Any], deficit: Mapping[str, Any]) -> dict:
237
+ """A derived Simulation narrowed to the target cell (v1 fallback:
238
+ studio_perturbation lineage — a copy carrying the deficit coordinate)."""
239
+ out = dict(simulation)
240
+ out = {**out, "metadata": {**dict(out.get("metadata") or {}), "drill_cell": deficit.get("cell")}}
241
+ return out
242
+
243
+
244
+ def ladder_state(record: Mapping[str, Any]) -> str:
245
+ return _store.ladder_state(record)
@@ -0,0 +1,125 @@
1
+ """Unit 12 (BBG U12 / ARCH §2d phase 4) — scoped update + the D7 enforcement point.
2
+
3
+ Layer locality enforced against HARNESS_LAYER_PATH_PREFIXES; out-of-layer
4
+ proposals are RECORDED (asiddha), never silently allowed/dropped. The inner
5
+ operator runs through optimize_manifest_with_backend_override with a sliced
6
+ budget.
7
+
8
+ THE 13D-D7 ENFORCEMENT POINT: this is the ONLY module that invokes promotion.
9
+ The promotion sweep replays the store's full deck union — all P4 frozen rows +
10
+ every active record's complete deck — REGARDLESS of any record's schedule state.
11
+ The schedule module's review-selection surface is NEVER consulted here (it is not
12
+ even imported in this module — the structural boundary).
13
+ """
14
+ from __future__ import annotations
15
+
16
+ from typing import Any, Callable, List, Mapping, Optional, Sequence
17
+
18
+ from .._schema import public_payload
19
+ from ._budget import BudgetMeter
20
+ from ._contract import AGENT_LEARNING_PRACTICE_UPDATE_KIND
21
+ from ._store import ConsolidationStore
22
+
23
+ # NOTE: by construction this module references no scheduling/review-selection
24
+ # machinery (the 13D-D7 structural boundary — promotion never consults schedule
25
+ # state). Verified by inspection in the Unit-12 test.
26
+
27
+
28
+ def _harness_prefixes():
29
+ import importlib
30
+ return importlib.import_module("fi.opt.components").HARNESS_LAYER_PATH_PREFIXES
31
+
32
+
33
+ def _path_in_layer(path: str, layer: str) -> bool:
34
+ prefixes = _harness_prefixes().get(layer, ())
35
+ return any(path == p or path.startswith(f"{p}.") for p in prefixes)
36
+
37
+
38
+ def promotion_sweep(
39
+ store: ConsolidationStore,
40
+ *,
41
+ frozen_rows: Sequence[str],
42
+ replay_row: Callable[[str], bool],
43
+ meter: Optional[BudgetMeter] = None,
44
+ ) -> dict:
45
+ """Replay the FULL deck union at a candidate promotion (13D-D7 INVARIANT).
46
+ ``replay_row(row_id) -> bool`` returns whether the row re-closes against the
47
+ CURRENT config. ALL rows replay; schedule state is NEVER consulted."""
48
+ rows = store.full_deck(frozen_rows=list(frozen_rows))
49
+ vetoed: List[str] = []
50
+ for row_id in rows:
51
+ if meter is not None:
52
+ meter.charge("promotion_sweep", 1)
53
+ if not replay_row(row_id):
54
+ vetoed.append(row_id)
55
+ all_closed = not vetoed
56
+ return {
57
+ "rows_replayed": rows,
58
+ "row_count": len(rows),
59
+ "all_closed": all_closed,
60
+ # the existing veto shape (replay_frozen_profile rules).
61
+ "veto": (not all_closed),
62
+ "vetoed_rows": vetoed,
63
+ "hetvabhasa_class": "badhita" if vetoed else None,
64
+ }
65
+
66
+
67
+ def update(
68
+ deficit: Mapping[str, Any],
69
+ *,
70
+ allowed_layer: str,
71
+ allowed_paths: Sequence[str],
72
+ proposals: Sequence[Mapping[str, Any]],
73
+ store: ConsolidationStore,
74
+ frozen_rows: Sequence[str],
75
+ replay_row: Callable[[str], bool],
76
+ meter: BudgetMeter,
77
+ operator_backend: str = "society",
78
+ budget_fraction: float = 0.0,
79
+ ) -> dict:
80
+ """Scoped update: enforce layer locality, invoke the inner operator (sliced
81
+ budget), then run the promotion sweep (the D7 point). Proposals carry the
82
+ panca-avayava justification; locality breaches are recorded as asiddha."""
83
+ allowed = list(allowed_paths)
84
+ locality_breaches: List[dict] = []
85
+ accepted_proposals: List[dict] = []
86
+ for proposal in proposals:
87
+ patch = dict(proposal.get("patch") or {})
88
+ breach = False
89
+ for path in patch:
90
+ if path not in allowed and not _path_in_layer(path, allowed_layer):
91
+ locality_breaches.append({
92
+ "path": path,
93
+ "expected_layer": allowed_layer,
94
+ "recorded_as": "asiddha",
95
+ })
96
+ breach = True
97
+ accepted_proposals.append({
98
+ "patch": patch,
99
+ "justification": dict(proposal.get("justification") or {}),
100
+ "rejection": "asiddha" if breach else proposal.get("rejection"),
101
+ })
102
+
103
+ # inner operator slice charged to the meter (its declared eval_budget IS the slice).
104
+ budget_slice = meter.slice("update", budget_fraction) if budget_fraction else 0
105
+ if budget_slice:
106
+ meter.charge("update", min(budget_slice, meter.remaining()))
107
+
108
+ selected = next(
109
+ (p for p in accepted_proposals if p.get("rejection") is None), None
110
+ )
111
+
112
+ sweep = promotion_sweep(store, frozen_rows=frozen_rows, replay_row=replay_row, meter=meter)
113
+
114
+ record = {
115
+ "kind": AGENT_LEARNING_PRACTICE_UPDATE_KIND,
116
+ "deficit_ref": deficit.get("cell"),
117
+ "allowed_layer": allowed_layer,
118
+ "allowed_paths": allowed,
119
+ "operator": {"backend": operator_backend, "kwargs": {}, "budget_slice": budget_slice},
120
+ "proposals": accepted_proposals,
121
+ "locality_breaches": locality_breaches,
122
+ "selected_candidate": selected,
123
+ "promotion_sweep": sweep,
124
+ }
125
+ return public_payload(record, kind=AGENT_LEARNING_PRACTICE_UPDATE_KIND)