agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,51 @@
1
+ """Phase 13D — the Practice Loop trainer (facade only; mirrors live/ style).
2
+
3
+ Lazy exports so ``import fi.alk.practice`` stays cheap. The trainer
4
+ employs the existing 13C operators; it adds no new step API and emits standard
5
+ ``agent-learning.run.v1`` rows through ``run_manifest``/``public_payload`` so
6
+ every episode lands a telemetry ledger row with zero new telemetry code.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import importlib
11
+ from typing import Any
12
+
13
+ # public name → home submodule (resolved lazily)
14
+ _LAZY_EXPORTS = {
15
+ # contract constants
16
+ "PRACTICE_PHASES": "_contract",
17
+ "PRACTICE_ARTIFACT_KINDS": "_contract",
18
+ "SCAFFOLD_TYPES": "_contract",
19
+ "LADDER_STATES": "_contract",
20
+ "PRACTICE_REPLAY_INTERVALS": "_contract",
21
+ "ZPD_BAND": "_contract",
22
+ "REVIEW_RATIO": "_contract",
23
+ "BUDGET_PLAN": "_contract",
24
+ "PRACTICE_STORE_ACTIVE_CAP": "_contract",
25
+ "SCAFFOLD_FADE_DEFAULT": "_contract",
26
+ "AGENT_LEARNING_PRACTICE_LOOP_KIND": "_contract",
27
+ "AGENT_LEARNING_PRACTICE_RESULT_KIND": "_contract",
28
+ "practice_store_path": "_contract",
29
+ # budget
30
+ "BudgetMeter": "_budget",
31
+ "BudgetExhausted": "_budget",
32
+ # trainer surface
33
+ "run_practice_loop": "_trainer",
34
+ "practice_report": "_assess",
35
+ "ladder_state": "_store",
36
+ "run_due_reviews": "_schedule",
37
+ }
38
+
39
+
40
+ def __getattr__(name: str) -> Any:
41
+ home = _LAZY_EXPORTS.get(name)
42
+ if home is None:
43
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
44
+ module = importlib.import_module(f"{__name__}.{home}")
45
+ value = getattr(module, name)
46
+ globals()[name] = value
47
+ return value
48
+
49
+
50
+ def __dir__() -> list[str]:
51
+ return sorted(set(globals()) | set(_LAZY_EXPORTS))
@@ -0,0 +1,103 @@
1
+ """Unit 10 (BBG U10 / ARCH §2d phase 1) — ASSESS: battery over the obligation grid.
2
+
3
+ Runs the battery over ``scenarios × cast × perturbations`` at ScenarioBinding
4
+ weights by deriving run manifests per cell (each scored episode charges the
5
+ meter), collecting verdict rows via loss.verdict_row, composing loss.loss_report.
6
+ Emits ``agent-learning.practice-report.v1`` through public_payload.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import Any, Callable, Dict, List, Mapping, Optional
11
+
12
+ from .._schema import public_payload
13
+ from .. import loss as _loss
14
+ from ._budget import BudgetMeter
15
+ from ._contract import AGENT_LEARNING_PRACTICE_REPORT_KIND
16
+
17
+
18
+ def _grid_cells(simulation: Mapping[str, Any]) -> List[dict]:
19
+ """Enumerate obligation cells from the P7 CoverageDeclaration vocabulary;
20
+ degenerate single cell when no coverage declared."""
21
+ cells: List[dict] = []
22
+ for binding in simulation.get("scenarios") or []:
23
+ scenario = binding.get("scenario") or {}
24
+ coverage = scenario.get("coverage") or {}
25
+ intents = coverage.get("intents") or [None]
26
+ perturbations = coverage.get("perturbations") or [None]
27
+ for member in binding.get("cast") or []:
28
+ for intent in intents:
29
+ for perturbation in perturbations:
30
+ cells.append({
31
+ "intent": intent,
32
+ "persona": member.get("persona"),
33
+ "perturbation": perturbation,
34
+ "obligation": None,
35
+ "weight": float(binding.get("weight", 1.0)),
36
+ })
37
+ if not cells:
38
+ cells.append({"intent": None, "persona": None, "perturbation": None,
39
+ "obligation": None, "weight": 1.0})
40
+ return cells
41
+
42
+
43
+ def assess(
44
+ simulation: Mapping[str, Any],
45
+ objective: Mapping[str, Any],
46
+ *,
47
+ meter: BudgetMeter,
48
+ round_no: int,
49
+ seed: int,
50
+ cell_scorer: Callable[[Mapping[str, Any]], Mapping[str, Any]],
51
+ parent_report_hash: Optional[str] = None,
52
+ repeats: int = 1,
53
+ coverage_source: str = "declared",
54
+ ) -> dict:
55
+ """Run the battery. ``cell_scorer(cell) -> {scalar, verdict, evidence_class}``
56
+ is the per-cell episode evaluator (injected for determinism/testing; in
57
+ production it derives + runs a run manifest). Each scored episode charges the
58
+ meter."""
59
+ cells = _grid_cells(simulation)
60
+ verdicts: List[dict] = []
61
+ calibration_mass_by_cell: Dict[str, float] = {}
62
+ for cell in cells:
63
+ for _ in range(max(1, int(repeats))):
64
+ meter.charge("assess", 1)
65
+ scored = cell_scorer(cell)
66
+ row = _loss.verdict_row(
67
+ eval_ref=scored.get("eval", "agent_report"),
68
+ cell=cell,
69
+ scalar=float(scored.get("scalar", 0.0)),
70
+ verdict=str(scored.get("verdict", "pass")),
71
+ evidence_class=str(scored.get("evidence_class", "local_gate")),
72
+ fidelity_admissible=bool(scored.get("fidelity_admissible", True)),
73
+ provenance={"round": round_no, "seed": seed},
74
+ )
75
+ verdicts.append(row)
76
+ if row["verdict"] == "unstable":
77
+ key = _loss._cell_key(cell)
78
+ calibration_mass_by_cell[key] = round(
79
+ calibration_mass_by_cell.get(key, 0.0) + 1.0, 6
80
+ )
81
+
82
+ loss_report = _loss.loss_report(objective, verdicts, budget_consumed=meter.consumed)
83
+ report = {
84
+ "kind": AGENT_LEARNING_PRACTICE_REPORT_KIND,
85
+ "round": int(round_no),
86
+ "objective_version": objective.get("version"),
87
+ "loss_report": loss_report,
88
+ "grid": {
89
+ "cells_total": len(cells),
90
+ "cells_assessed": len(cells),
91
+ "coverage_source": coverage_source,
92
+ },
93
+ "calibration_mass_by_cell": calibration_mass_by_cell,
94
+ "budget_consumed": meter.consumed,
95
+ "seed": int(seed),
96
+ "parent": parent_report_hash,
97
+ }
98
+ return public_payload(report, kind=AGENT_LEARNING_PRACTICE_REPORT_KIND)
99
+
100
+
101
+ def practice_report(*args: Any, **kwargs: Any) -> dict:
102
+ """Public alias for the ASSESS report builder (facade export)."""
103
+ return assess(*args, **kwargs)
@@ -0,0 +1,81 @@
1
+ """Unit 8 (BBG U8 / ARCH §2d, AD-I) — the single budget meter.
2
+
3
+ ONE unit = one scored episode evaluation. Every assess row, ZPD repeat,
4
+ scaffolded/unscaffolded drill evaluation, inner-operator evaluation, scheduled
5
+ review row, and promotion-sweep row charges THIS meter — there is no second
6
+ currency. Soft per-phase enforcement of budget_plan with carry-over.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import Dict
11
+
12
+ from ._contract import BUDGET_PLAN, PRACTICE_PHASES
13
+
14
+ # Map the 4-fraction budget_plan onto phases. assess / drill / update / review;
15
+ # diagnose+consolidate+calibrate draw from their adjacent phase allocations.
16
+ _BUDGET_PLAN_PHASES = ("assess", "drill", "update", "review")
17
+
18
+
19
+ class BudgetExhausted(RuntimeError):
20
+ """Raised when the meter has no remaining budget (trainer stop)."""
21
+
22
+
23
+ class BudgetMeter:
24
+ """The single eval-unit meter (AD-I)."""
25
+
26
+ def __init__(self, total: int, *, budget_plan: tuple[float, ...] = BUDGET_PLAN) -> None:
27
+ if not isinstance(total, int) or isinstance(total, bool) or total < 1:
28
+ raise ValueError("budget total must be an int >= 1")
29
+ self.total = int(total)
30
+ self.consumed = 0
31
+ self._by_phase: Dict[str, int] = {}
32
+ self._plan = tuple(budget_plan)
33
+ # per-phase soft caps (allocation of total) keyed by the 4 plan phases.
34
+ self._caps = {
35
+ phase: int(round(self.total * frac))
36
+ for phase, frac in zip(_BUDGET_PLAN_PHASES, self._plan)
37
+ }
38
+
39
+ def _plan_phase(self, phase: str) -> str:
40
+ if phase in _BUDGET_PLAN_PHASES:
41
+ return phase
42
+ if phase == "diagnose":
43
+ return "assess"
44
+ if phase in ("consolidate", "calibrate"):
45
+ return "update"
46
+ return "drill"
47
+
48
+ def charge(self, phase: str, n: int = 1) -> int:
49
+ if phase not in PRACTICE_PHASES and phase not in ("review", "promotion_sweep"):
50
+ raise ValueError(f"unknown budget phase {phase!r}")
51
+ if n < 0:
52
+ raise ValueError("charge n must be >= 0")
53
+ if self.consumed + n > self.total:
54
+ raise BudgetExhausted(
55
+ f"budget exhausted: consumed={self.consumed} + {n} > total={self.total}"
56
+ )
57
+ self.consumed += n
58
+ self._by_phase[phase] = self._by_phase.get(phase, 0) + n
59
+ return self.consumed
60
+
61
+ def remaining(self) -> int:
62
+ return self.total - self.consumed
63
+
64
+ def slice(self, phase: str, fraction: float) -> int:
65
+ """Return an integer sub-budget handed to inner operators (their declared
66
+ eval_budget IS the slice). Bounded by remaining budget."""
67
+ if not 0.0 <= fraction <= 1.0:
68
+ raise ValueError("slice fraction must be in [0, 1]")
69
+ want = int(self.total * fraction)
70
+ return max(0, min(want, self.remaining()))
71
+
72
+ def ledger(self) -> dict:
73
+ """Per-phase consumption; conservation: sum(phase) == consumed <= total."""
74
+ by_phase = {p: self._by_phase.get(p, 0) for p in sorted(self._by_phase)}
75
+ assert sum(by_phase.values()) == self.consumed <= self.total
76
+ return {
77
+ "total": self.total,
78
+ "consumed": self.consumed,
79
+ "remaining": self.remaining(),
80
+ "by_phase": by_phase,
81
+ }
@@ -0,0 +1,69 @@
1
+ """Unit 13 (BBG U13 / ARCH §2d phase 6) — CALIBRATE: the learned-gate.
2
+
3
+ Per cell: learned iff score ≥ floor AND fork-entropy ≤ threshold AND ICC ≥ floor
4
+ over k; high-score/high-entropy = fluent_not_learned (stays in rotation);
5
+ plateaued/zpd_exited stop rules. Trajectory profiles are post-hoc, never a stop
6
+ rule. Emits ``agent-learning.practice-calibration.v1``.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import Any, Mapping, Optional, Sequence
11
+
12
+ from .._schema import public_payload
13
+ from ..live._contract import UNSTABLE_ICC_FLOOR
14
+ from ._contract import AGENT_LEARNING_PRACTICE_CALIBRATION_KIND, CALIBRATION_VERDICTS
15
+
16
+
17
+ def calibrate_cell(
18
+ cell: Mapping[str, Any],
19
+ *,
20
+ score: float,
21
+ fork_entropy: float,
22
+ divergence_step: Optional[int],
23
+ icc: float,
24
+ repeats: int,
25
+ score_floor: float = 0.7,
26
+ entropy_threshold: float = 0.3,
27
+ icc_floor: float = UNSTABLE_ICC_FLOOR,
28
+ prior_score: Optional[float] = None,
29
+ in_band: bool = True,
30
+ ) -> dict:
31
+ """Compute one cell's calibration verdict (synthesis §4(6))."""
32
+ learned = score >= score_floor and fork_entropy <= entropy_threshold and icc >= icc_floor
33
+ if learned:
34
+ verdict = "learned"
35
+ stop_reason = "learned"
36
+ elif score >= score_floor and fork_entropy > entropy_threshold:
37
+ verdict = "fluent_not_learned" # high-score / high-entropy
38
+ stop_reason = None
39
+ elif not in_band:
40
+ verdict = "zpd_exited"
41
+ stop_reason = "zpd_exited"
42
+ elif prior_score is not None and abs(score - prior_score) < 1e-3:
43
+ verdict = "plateaued"
44
+ stop_reason = "plateaued"
45
+ else:
46
+ verdict = "in_rotation"
47
+ stop_reason = None
48
+ assert verdict in CALIBRATION_VERDICTS
49
+ return {
50
+ "cell": dict(cell),
51
+ "score": round(float(score), 6),
52
+ "fork_entropy": round(float(fork_entropy), 6),
53
+ "divergence_step": divergence_step,
54
+ "icc": round(float(icc), 6),
55
+ "repeats": int(repeats),
56
+ "verdict": verdict,
57
+ "stop_reason": stop_reason,
58
+ }
59
+
60
+
61
+ def calibrate(cells: Sequence[Mapping[str, Any]], *, round_no: int) -> dict:
62
+ """Emit the calibration artifact over a list of pre-computed cell measures."""
63
+ records = [calibrate_cell(**c) if "verdict" not in c else dict(c) for c in cells]
64
+ report = {
65
+ "kind": AGENT_LEARNING_PRACTICE_CALIBRATION_KIND,
66
+ "round": int(round_no),
67
+ "cells": records,
68
+ }
69
+ return public_payload(report, kind=AGENT_LEARNING_PRACTICE_CALIBRATION_KIND)
@@ -0,0 +1,86 @@
1
+ """Unit 22 (BBG U22 / RU-7) — the capstone A/B harness.
2
+
3
+ An EXPERIMENT, not a release gate (gates stay deterministic; nothing here
4
+ registers a check). The harness runs the practice loop vs real search backends
5
+ at EQUAL TOTAL metered budget (the one meter, AD-I) over kit-local fixtures, and
6
+ REFUSES to print a headline unless every arm completed the same declared total
7
+ (``headline: null`` + ``ab_budget_mismatch`` otherwise — doctrine #11).
8
+
9
+ This module builds the harness so it CAN run offline-deterministically; running
10
+ the capstone experiment + writing the paper is a separate later task.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ from pathlib import Path
16
+ from typing import List
17
+
18
+ from .._schema import public_payload
19
+ from ._contract import AGENT_LEARNING_PRACTICE_LOOP_KIND
20
+
21
+ # RU-7: real backend tokens only (the canon tuple stays closed). "greedy" = bandit.
22
+ CAPSTONE_ARMS = ("practice_loop", "gepa", "tpe", "society", "bandit")
23
+ # manifest-level ablation knobs of the practice arm (never a code fork).
24
+ CAPSTONE_ABLATIONS = ("a1_no_zpd", "a2_no_spacing", "a3_no_consolidation", "a4_no_calibration")
25
+
26
+
27
+ def _load_config(manifest_dir: Path) -> dict:
28
+ config_path = manifest_dir / "capstone.json"
29
+ if not config_path.exists():
30
+ raise FileNotFoundError(f"capstone config not found at {config_path}")
31
+ return json.loads(config_path.read_text())
32
+
33
+
34
+ def run_ab(manifest_dir: str | Path) -> dict:
35
+ """Run the A/B harness. Reads ``capstone.json`` declaring the arms and the
36
+ equal total budget; enforces the equal-budget headline rule (doctrine #11).
37
+
38
+ The arm execution is offline-deterministic: each arm reports its declared
39
+ total metered budget and a (placeholder until the experiment runs)
40
+ retention_after_interference. Running the experiment itself is a later task;
41
+ this harness validates the equal-budget contract and emits the ab_harness
42
+ block."""
43
+ manifest_dir = Path(manifest_dir)
44
+ config = _load_config(manifest_dir)
45
+ declared_total = int(config.get("eval_budget", 0))
46
+ arms_decl = config.get("arms") or list(CAPSTONE_ARMS)
47
+
48
+ arms: List[dict] = []
49
+ budgets: set[int] = set()
50
+ for arm in arms_decl:
51
+ arm_total = int(config.get("arm_budgets", {}).get(arm, declared_total))
52
+ budgets.add(arm_total)
53
+ arms.append({
54
+ "arm": arm,
55
+ "total_metered_budget": arm_total,
56
+ # best_found is printed per arm precisely so a search arm may visibly
57
+ # win best-found while losing retention (the headline).
58
+ "best_found": None,
59
+ "retention_after_interference": None,
60
+ })
61
+
62
+ # equal TOTAL metered budget per arm (AD-I) — else headline null + warning.
63
+ budget_match = len(budgets) == 1 and declared_total in budgets
64
+ findings: List[dict] = []
65
+ headline = None
66
+ if not budget_match:
67
+ findings.append({
68
+ "type": "ab_budget_mismatch", "level": "warning",
69
+ "reason": f"arms did not complete the same declared total ({sorted(budgets)} != {declared_total})",
70
+ })
71
+ else:
72
+ headline = {"metric": "retention_after_interference", "by_arm": None,
73
+ "note": "populated when the experiment runs (a later task)"}
74
+
75
+ payload = {
76
+ "kind": AGENT_LEARNING_PRACTICE_LOOP_KIND,
77
+ "ab_harness": {
78
+ "arms": arms,
79
+ "ablations": list(CAPSTONE_ABLATIONS),
80
+ "equal_total_budget": declared_total,
81
+ "budget_match": budget_match,
82
+ "headline": headline,
83
+ "findings": findings,
84
+ },
85
+ }
86
+ return public_payload(payload, kind=AGENT_LEARNING_PRACTICE_LOOP_KIND)["ab_harness"]
@@ -0,0 +1,91 @@
1
+ """Unit 8 (BBG U8 / ARCH §3) — practice vocabularies + canon constants.
2
+
3
+ Every constant ARCH §3 freezes for the Practice Loop, verbatim. RU-1 numeric
4
+ defaults. Evidence/verdict vocab is IMPORTED from live/_contract.py, never
5
+ redeclared (ARCH §1.7). The Unit-20 gate byte-compares these.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ from pathlib import Path
11
+
12
+ from ..live._contract import ( # noqa: F401 (re-exported canon)
13
+ DEFAULT_REPEATS,
14
+ EVIDENCE_CLASSES,
15
+ RELEASE_ADMISSIBLE_EVIDENCE_CLASSES,
16
+ UNSTABLE_ICC_FLOOR,
17
+ VERDICTS,
18
+ )
19
+ from ..loss import ( # noqa: F401 (re-export the objective/loss-report kinds)
20
+ AGENT_LEARNING_LOSS_REPORT_KIND,
21
+ AGENT_LEARNING_OBJECTIVE_KIND,
22
+ )
23
+
24
+ # --- artifact kinds (RU-4) -------------------------------------------------
25
+ AGENT_LEARNING_PRACTICE_LOOP_KIND = "agent-learning.practice-loop.v1"
26
+ AGENT_LEARNING_PRACTICE_RESULT_KIND = "agent-learning.practice-result.v1"
27
+ AGENT_LEARNING_PRACTICE_REPORT_KIND = "agent-learning.practice-report.v1"
28
+ AGENT_LEARNING_PRACTICE_DEFICITS_KIND = "agent-learning.practice-deficits.v1"
29
+ AGENT_LEARNING_PRACTICE_DRILL_KIND = "agent-learning.practice-drill.v1"
30
+ AGENT_LEARNING_PRACTICE_UPDATE_KIND = "agent-learning.practice-update.v1"
31
+ AGENT_LEARNING_CONSOLIDATED_LESSON_KIND = "agent-learning.consolidated-lesson.v1"
32
+ AGENT_LEARNING_PRACTICE_CALIBRATION_KIND = "agent-learning.practice-calibration.v1"
33
+
34
+ PRACTICE_ARTIFACT_KINDS = (
35
+ AGENT_LEARNING_PRACTICE_LOOP_KIND,
36
+ AGENT_LEARNING_PRACTICE_RESULT_KIND,
37
+ AGENT_LEARNING_PRACTICE_REPORT_KIND,
38
+ AGENT_LEARNING_PRACTICE_DEFICITS_KIND,
39
+ AGENT_LEARNING_PRACTICE_DRILL_KIND,
40
+ AGENT_LEARNING_PRACTICE_UPDATE_KIND,
41
+ AGENT_LEARNING_CONSOLIDATED_LESSON_KIND,
42
+ AGENT_LEARNING_PRACTICE_CALIBRATION_KIND,
43
+ )
44
+
45
+ # --- phases + vocabularies (ARCH §3) ---------------------------------------
46
+ PRACTICE_PHASES = ("assess", "diagnose", "drill", "update", "consolidate", "calibrate")
47
+ SCAFFOLD_TYPES = ("world_simplification", "hint_tool", "worked_example", "relaxed_success")
48
+ ZPD_VERDICTS = ("in_band", "vygotsky_form", "below_band", "above_band", "unstable")
49
+ CALIBRATION_VERDICTS = ("learned", "fluent_not_learned", "in_rotation", "plateaued", "zpd_exited")
50
+ LADDER_STATES = ("episodic", "instruction", "skill")
51
+ PRACTICE_REPLAY_INTERVALS = (1, 2, 4, 8, 16) # cap 16
52
+ STORE_STATUSES = ("active", "retired")
53
+ RETIREMENT_REASONS = ("repeated_failure", "obsolete")
54
+ LESSON_KINDS = ("instruction_block", "config_patch", "skill")
55
+
56
+ # --- 13D-5 capstone ablation knobs (additive; the experiment path only) -----
57
+ # Real trainer config flags that change run_practice_loop behaviour (never
58
+ # labels): A1 disables ZPD filtering, A2 disables standing spaced reviews
59
+ # (replay only at promotion), A3 skips the consolidate phase entirely, A4
60
+ # disables the calibration learned-gate (fixed-k, never stop early).
61
+ PRACTICE_ABLATIONS = ("a1_no_zpd", "a2_no_spacing", "a3_no_consolidation", "a4_no_calibration")
62
+
63
+ # --- RU-1 defaults ---------------------------------------------------------
64
+ ZPD_BAND = (0.2, 0.7)
65
+ REVIEW_RATIO = 0.25
66
+ BUDGET_PLAN = (0.25, 0.35, 0.25, 0.15) # assess / drill / update / review
67
+ PRACTICE_STORE_ACTIVE_CAP = 64
68
+ SCAFFOLD_FADE_DEFAULT = (1.0, 0.5, 0.0) # MUST end at 0.0
69
+ MAX_REPLAY_INTERVAL = 16
70
+ DEFAULT_MAX_ROUNDS = 8
71
+ DEFAULT_INNER_OPERATOR_BACKEND = "society"
72
+
73
+ # --- store placement (AD-G — the Phase-8 ledger precedent) -----------------
74
+ LESSON_ID_PREFIX = "lesson_"
75
+ PRACTICE_STORE_PATH_ENV = "AGENT_LEARNING_PRACTICE_STORE_PATH"
76
+ PRACTICE_STORE_HOME_ENV = "AGENT_LEARNING_HOME"
77
+ PRACTICE_STORE_DIR_NAME = "practice"
78
+ PRACTICE_STORE_FILE_NAME = "records.jsonl"
79
+
80
+
81
+ def practice_store_path(override: str | Path | None = None) -> Path:
82
+ """Resolve the consolidation store path (AD-G). Precedence: explicit arg >
83
+ AGENT_LEARNING_PRACTICE_STORE_PATH > ${AGENT_LEARNING_HOME:-~/.agent-learning}
84
+ /practice/records.jsonl."""
85
+ if override is not None:
86
+ return Path(override)
87
+ env_override = os.environ.get(PRACTICE_STORE_PATH_ENV)
88
+ if env_override:
89
+ return Path(env_override)
90
+ home = os.environ.get(PRACTICE_STORE_HOME_ENV) or (Path.home() / ".agent-learning")
91
+ return Path(home) / PRACTICE_STORE_DIR_NAME / PRACTICE_STORE_FILE_NAME
@@ -0,0 +1,79 @@
1
+ """Unit 10 (BBG U10 / ARCH §2d phase 2) — DIAGNOSE: pure composition.
2
+
3
+ Ranks weak cells, attributes each to a harness_layer ∈ HARNESS_LAYERS via
4
+ ComponentDiagnosis and relevant_search_paths (narrowing, never widening). Credit
5
+ method "layer_scoped" always; "counterfactual_replay" (13C T7) only budget-
6
+ permitting (fallback = layer scoping only). Emits
7
+ ``agent-learning.practice-deficits.v1`` ranked deterministically (loss desc,
8
+ tie-break by cell content hash). No new machinery.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ from typing import Any, List, Mapping, Optional
14
+
15
+ from .._schema import public_payload
16
+ from ._contract import AGENT_LEARNING_PRACTICE_DEFICITS_KIND
17
+
18
+
19
+ def _components():
20
+ import importlib
21
+ return importlib.import_module("fi.opt.components")
22
+
23
+
24
+ def _cell_hash(cell: Mapping[str, Any]) -> str:
25
+ return json.dumps(cell, sort_keys=True, default=str)
26
+
27
+
28
+ def diagnose(
29
+ practice_report: Mapping[str, Any],
30
+ *,
31
+ search_space: Mapping[str, Any],
32
+ layer_hint: Optional[Mapping[str, str]] = None,
33
+ allow_counterfactual: bool = False,
34
+ ) -> dict:
35
+ """Pure composition over the ASSESS report's loss cells. ``layer_hint`` maps a
36
+ cell key → harness_layer (from upstream diagnosis); default 'execution'."""
37
+ components = _components()
38
+ harness_layers = components.HARNESS_LAYERS
39
+ prefixes = components.HARNESS_LAYER_PATH_PREFIXES
40
+ layer_hint = dict(layer_hint or {})
41
+
42
+ loss_report = practice_report.get("loss_report") or {}
43
+ cells = loss_report.get("cells") or []
44
+ # rank weak cells: loss desc, tie-break by cell content hash.
45
+ ranked = sorted(
46
+ cells,
47
+ key=lambda c: (-float(c.get("loss", 0.0)), _cell_hash(c.get("cell") or {})),
48
+ )
49
+
50
+ deficits: List[dict] = []
51
+ for cell_report in ranked:
52
+ cell = cell_report.get("cell") or {}
53
+ if float(cell_report.get("loss", 0.0)) <= 0.0:
54
+ continue # closed cells are not deficits
55
+ layer = layer_hint.get(_cell_hash(cell), "execution")
56
+ if layer not in harness_layers:
57
+ layer = "execution"
58
+ # narrowing search paths from the layer's prefixes.
59
+ layer_prefixes = prefixes.get(layer, ())
60
+ narrowed = sorted(
61
+ path for path in search_space
62
+ if any(path == p or path.startswith(f"{p}.") for p in layer_prefixes)
63
+ )
64
+ method = "counterfactual_replay" if allow_counterfactual else "layer_scoped"
65
+ deficits.append({
66
+ "cell": cell,
67
+ "harness_layer": layer,
68
+ "search_paths": narrowed,
69
+ "credit": {"method": method, "rows": []},
70
+ "evidence_rows": cell_report.get("verdicts") or [],
71
+ })
72
+
73
+ report = {
74
+ "kind": AGENT_LEARNING_PRACTICE_DEFICITS_KIND,
75
+ "round": practice_report.get("round"),
76
+ "objective_version": practice_report.get("objective_version"),
77
+ "deficits": deficits,
78
+ }
79
+ return public_payload(report, kind=AGENT_LEARNING_PRACTICE_DEFICITS_KIND)