agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/live/_stats.py ADDED
@@ -0,0 +1,561 @@
1
+ """Variance math, repeat executor, verdicts for live lanes — pure numpy.
2
+
3
+ ICC(1) via one-way variance decomposition (ARCH Decision 4 — no scipy);
4
+ degenerate zero-variance matrices define ICC := 1.0 (a deterministic green
5
+ run must never classify ``unstable``). Determinism metrics are reported
6
+ separately from quality scores — DFAH's r=-0.11 forbids conflation (R§1 #15).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import dataclasses
12
+ import math
13
+ import tempfile
14
+ import time
15
+ import uuid
16
+ from pathlib import Path
17
+ from typing import Any, Callable, Mapping, Sequence
18
+
19
+ import numpy as np
20
+
21
+ from .._schema import public_payload
22
+ from ._contract import (
23
+ AGENT_LEARNING_RUN_KIND,
24
+ DEFAULT_REPEATS,
25
+ EVIDENCE_CLASSES,
26
+ FAILURE_LAYERS,
27
+ UNSTABLE_ICC_FLOOR,
28
+ LaneRun,
29
+ lane_budget_s,
30
+ )
31
+ from ._transcript import TranscriptRecorder
32
+
33
+
34
+ def icc_and_within_variance(scores: np.ndarray) -> tuple[float, float]:
35
+ """One-way random-effects ICC over a (n_scenarios, k_repeats) score
36
+ matrix + pooled within-scenario variance (R§1 #16, ICC convergence at
37
+ n=8–16; P3-D2 default k=8).
38
+
39
+ ICC = (MS_between - MS_within) / (MS_between + (k-1)·MS_within)
40
+ """
41
+
42
+ scores = np.asarray(scores, dtype=float)
43
+ if scores.ndim != 2:
44
+ raise ValueError("scores must be a 2-D (n_scenarios, k_repeats) matrix")
45
+ n, k = scores.shape
46
+ grand = scores.mean()
47
+ row_means = scores.mean(axis=1)
48
+ ms_between = k * ((row_means - grand) ** 2).sum() / max(n - 1, 1)
49
+ ms_within = ((scores - row_means[:, None]) ** 2).sum() / max(n * (k - 1), 1)
50
+ denominator = ms_between + (k - 1) * ms_within
51
+ if denominator == 0:
52
+ # Degenerate zero-variance matrix (e.g. an all-pass run with
53
+ # byte-identical scores): perfect consistency by definition —
54
+ # ICC := 1.0. Without this rule a perfectly deterministic green
55
+ # run would classify `unstable` (review finding F2).
56
+ return 1.0, float(ms_within)
57
+ return float((ms_between - ms_within) / denominator), float(ms_within)
58
+
59
+
60
+ def divergence_step(step_signatures: Sequence[Sequence[str]]) -> int | None:
61
+ """First step index at which repeated trajectories fork (R§1 #17 — 69%
62
+ fork at step 2, so this is cheap, high-signal evidence). Signatures are
63
+ normalized step strings (tool name + outcome class, no payloads).
64
+ Returns None when all repeats share one trajectory."""
65
+
66
+ longest = max((len(s) for s in step_signatures), default=0)
67
+ for index in range(longest):
68
+ prefixes = {tuple(s[: index + 1]) for s in step_signatures}
69
+ if len(prefixes) > 1:
70
+ return index
71
+ return None
72
+
73
+
74
+ def determinism_metrics(
75
+ step_signatures: Sequence[Sequence[str]],
76
+ ) -> dict[str, Any]:
77
+ """Trajectory-distribution metrics, kept strictly separate from quality
78
+ (R§1 #15). Entropy is Shannon entropy in bits over distinct trajectories."""
79
+
80
+ trajectories = [tuple(signature) for signature in step_signatures]
81
+ if not trajectories:
82
+ return {"distinct_trajectory_count": 0, "trajectory_entropy": 0.0}
83
+ counts: dict[tuple[str, ...], int] = {}
84
+ for trajectory in trajectories:
85
+ counts[trajectory] = counts.get(trajectory, 0) + 1
86
+ total = len(trajectories)
87
+ entropy = -sum(
88
+ (count / total) * math.log2(count / total) for count in counts.values()
89
+ )
90
+ return {
91
+ "distinct_trajectory_count": len(counts),
92
+ "trajectory_entropy": round(float(entropy), 6),
93
+ }
94
+
95
+
96
+ def step_signature_from_events(
97
+ events: Sequence[Mapping[str, Any]],
98
+ ) -> list[str]:
99
+ """Normalize a transcript into step strings for divergence detection:
100
+ tool name + outcome class, message marks — never payloads."""
101
+
102
+ signature: list[str] = []
103
+ for event in events:
104
+ channel = str(event.get("channel") or "")
105
+ event_type = str(event.get("type") or "")
106
+ payload = event.get("payload")
107
+ payload = payload if isinstance(payload, Mapping) else {}
108
+ if channel == "tool":
109
+ name = str(payload.get("name") or payload.get("tool") or "tool")
110
+ if payload.get("error") or payload.get("ok") is False:
111
+ outcome = "error"
112
+ else:
113
+ outcome = "ok"
114
+ signature.append(f"tool:{name}:{outcome}")
115
+ elif channel in ("user", "agent"):
116
+ signature.append(f"{channel}:{event_type}")
117
+ return signature
118
+
119
+
120
+ @dataclasses.dataclass
121
+ class LaneRunResult:
122
+ lane: str
123
+ evidence_class: str # stamped at construction; member of EVIDENCE_CLASSES
124
+ repeats: int
125
+ verdict: str # "pass" | "fail" | "unstable" | "void"
126
+ per_repeat: list[dict] # {score, passed, failure_layer, transcript_path}
127
+ icc: float | None
128
+ within_variance: float | None
129
+ divergence_step: int | None
130
+ determinism: dict # {distinct_trajectory_count, trajectory_entropy}
131
+ quarantined_repeats: int # lane_infra rows excluded from stats (R§1 #7 validate-then-score)
132
+ required_env: list[str] # NAMES only, never values
133
+ end_state_diff: dict | None # before/after snapshot (R§1 #14 Saber)
134
+ # --- run identity + budget mechanics (ARCH §4) — open details, simplest
135
+ # additive fields consistent with the architecture: ---------------------
136
+ run_id: str = ""
137
+ rung: str | int = 1
138
+ framework: str | None = None
139
+ framework_version: str | None = None
140
+ version_requirement: str | None = None
141
+ version_ok: bool | None = None
142
+ repeats_requested: int = 0
143
+ repeats_completed: int = 0
144
+ budget_cap_s: float = 0.0
145
+ budget_spent_s: float = 0.0
146
+ verdict_reason: str | None = None
147
+ findings: list[dict] = dataclasses.field(default_factory=list)
148
+ artifacts_dir: str | None = None
149
+
150
+ def to_block(self) -> dict[str, Any]:
151
+ """The ``live_lane`` evidence block of the run.v1 payload."""
152
+
153
+ return dataclasses.asdict(self)
154
+
155
+
156
+ def run_repeated(
157
+ run_once: Callable[[int, TranscriptRecorder], dict],
158
+ *,
159
+ lane: str,
160
+ evidence_class: str,
161
+ repeats: int = DEFAULT_REPEATS, # P3-D2 default; --repeats override upstream
162
+ budget_s: float | None = None, # None → LANE_BUDGET_S.get(lane, default):
163
+ # 600 s default, 900 s voice lanes (P3-D2)
164
+ unstable_icc_floor: float = UNSTABLE_ICC_FLOOR,
165
+ required_env: Sequence[str] = (),
166
+ artifacts_dir: str | Path | None = None,
167
+ run_id: str | None = None,
168
+ rung: str | int = 1,
169
+ framework: str | None = None,
170
+ version_requirement: str | None = None,
171
+ ) -> LaneRunResult:
172
+ """Repeat executor. lane_infra rows are quarantined (excluded from the
173
+ score matrix AND counted); verifier evidence is mandatory per repeat —
174
+ a repeat with no programmatic/judge/end-state verdict is itself
175
+ lane_infra (R§1 #5: sampling without verification is the documented gap).
176
+
177
+ Per-scenario verdict (R§3.4; lane-run exit policy lives in unit 6):
178
+ pass — every non-quarantined repeat passed AND icc >= floor
179
+ (zero-variance all-pass runs hit this via ICC := 1.0 above)
180
+ fail — every non-quarantined repeat failed
181
+ unstable — mixed outcomes, or icc < floor; quarantined like a flaky
182
+ test with fork evidence attached, never a red/green coin flip
183
+ void — lane_infra consumed the sample (no scoreable repeats);
184
+ the ONLY source of `void` (PRD §4.1).
185
+ """
186
+
187
+ if evidence_class not in EVIDENCE_CLASSES:
188
+ raise ValueError(f"unknown evidence_class: {evidence_class!r}")
189
+ if repeats < 1:
190
+ raise ValueError("repeats must be >= 1")
191
+ budget = float(budget_s) if budget_s is not None else lane_budget_s(lane)
192
+ base_dir = (
193
+ Path(artifacts_dir)
194
+ if artifacts_dir is not None
195
+ else Path(tempfile.mkdtemp(prefix=f"agent-learning-live-{lane}-"))
196
+ )
197
+ base_dir.mkdir(parents=True, exist_ok=True)
198
+ resolved_run_id = run_id or uuid.uuid4().hex
199
+ started = time.monotonic()
200
+
201
+ rows: list[LaneRun] = []
202
+ findings: list[dict] = []
203
+ framework_name = framework
204
+ framework_version: str | None = None
205
+ version_ok_observed: bool | None = None
206
+ end_state_diff: dict | None = None
207
+ budget_exhausted = False
208
+
209
+ for index in range(repeats):
210
+ if time.monotonic() - started >= budget:
211
+ budget_exhausted = True
212
+ break
213
+ transcript = TranscriptRecorder(
214
+ base_dir / f"repeat-{index:02d}.jsonl",
215
+ required_env=required_env,
216
+ )
217
+ try:
218
+ outcome: Mapping[str, Any] = run_once(index, transcript) or {}
219
+ except Exception as exc: # our machinery failing is lane_infra, never a score
220
+ outcome = {
221
+ "passed": None,
222
+ "score": None,
223
+ "failure_layer": "lane_infra",
224
+ "void_reason": f"lane runner exception: {exc}",
225
+ "detail": f"lane runner exception: {exc}",
226
+ }
227
+ finally:
228
+ summary = transcript.close()
229
+
230
+ failure_layer = outcome.get("failure_layer")
231
+ if failure_layer is not None and failure_layer not in FAILURE_LAYERS:
232
+ failure_layer = "lane_infra"
233
+ passed = outcome.get("passed")
234
+ if passed is None and failure_layer is None:
235
+ # No verdict at all → the repeat itself is lane_infra (R§1 #5).
236
+ failure_layer = "lane_infra"
237
+ outcome = dict(outcome)
238
+ outcome.setdefault(
239
+ "void_reason", "no verifier evidence for this repeat"
240
+ )
241
+ outcome.setdefault(
242
+ "detail", "no verifier evidence for this repeat"
243
+ )
244
+ quarantined = failure_layer == "lane_infra"
245
+ score = outcome.get("score")
246
+ if score is None and not quarantined:
247
+ score = 1.0 if passed else 0.0
248
+
249
+ version_info = outcome.get("version")
250
+ if isinstance(version_info, Mapping):
251
+ framework_name = framework_name or version_info.get("framework")
252
+ framework_version = framework_version or version_info.get(
253
+ "framework_version"
254
+ )
255
+ if version_info.get("version_ok") is False:
256
+ version_ok_observed = False
257
+ findings.append(
258
+ {
259
+ "type": "live_lane_framework_version_mismatch",
260
+ "level": "error",
261
+ "repeat": index,
262
+ "detail": version_info.get("void_reason"),
263
+ }
264
+ )
265
+ elif version_ok_observed is None:
266
+ version_ok_observed = bool(version_info.get("version_ok"))
267
+ if isinstance(outcome.get("end_state_diff"), Mapping):
268
+ end_state_diff = dict(outcome["end_state_diff"])
269
+
270
+ if not summary.get("complete", True):
271
+ findings.append(
272
+ {
273
+ "type": "live_lane_transcript_truncated",
274
+ "level": "warning",
275
+ "repeat": index,
276
+ "detail": summary.get("truncated"),
277
+ }
278
+ )
279
+
280
+ rows.append(
281
+ LaneRun(
282
+ index=index,
283
+ passed=None if quarantined else bool(passed),
284
+ score=None if quarantined else float(score),
285
+ failure_layer=failure_layer,
286
+ quarantined=quarantined,
287
+ evidence_class=evidence_class,
288
+ detail=str(outcome.get("detail") or ""),
289
+ void_reason=outcome.get("void_reason"),
290
+ transcript_path=str(summary.get("path")),
291
+ transcript_complete=bool(summary.get("complete", True)),
292
+ transcript_sha256=summary.get("sha256"),
293
+ step_signature=tuple(outcome.get("step_signature") or ()),
294
+ )
295
+ )
296
+
297
+ budget_spent = time.monotonic() - started
298
+ scoreable = [row for row in rows if not row.quarantined]
299
+ quarantined_count = sum(1 for row in rows if row.quarantined)
300
+
301
+ if scoreable:
302
+ matrix = np.asarray([[row.score for row in scoreable]], dtype=float)
303
+ icc, within = icc_and_within_variance(matrix)
304
+ else:
305
+ icc, within = None, None
306
+
307
+ signatures = [
308
+ list(row.step_signature) for row in scoreable if row.step_signature
309
+ ]
310
+ fork_step = divergence_step(signatures) if signatures else None
311
+ determinism = determinism_metrics(signatures)
312
+
313
+ verdict_reason: str | None = None
314
+ if not scoreable:
315
+ verdict = "void"
316
+ verdict_reason = "lane_infra_consumed_sample"
317
+ findings.append(
318
+ {
319
+ "type": "live_lane_infra_void",
320
+ "level": "error",
321
+ "detail": "lane_infra consumed the sample (no scoreable repeats)",
322
+ }
323
+ )
324
+ elif all(row.passed for row in scoreable):
325
+ if icc is not None and icc < unstable_icc_floor:
326
+ verdict = "unstable"
327
+ verdict_reason = "icc_below_floor"
328
+ else:
329
+ verdict = "pass"
330
+ elif all(not row.passed for row in scoreable):
331
+ verdict = "fail"
332
+ else:
333
+ verdict = "unstable"
334
+ verdict_reason = "mixed_outcomes"
335
+
336
+ if budget_exhausted and verdict == "pass":
337
+ # Hitting a cap mid-run yields `unstable` with reason budget_exhausted
338
+ # rather than a silently smaller n (ARCH §4 budget mechanics).
339
+ verdict = "unstable"
340
+ verdict_reason = "budget_exhausted"
341
+
342
+ if verdict == "unstable":
343
+ findings.append(
344
+ {
345
+ "type": "live_lane_scenario_unstable",
346
+ "level": "warning",
347
+ "detail": {
348
+ "reason": verdict_reason,
349
+ "icc": icc,
350
+ "divergence_step": fork_step,
351
+ },
352
+ }
353
+ )
354
+
355
+ return LaneRunResult(
356
+ lane=lane,
357
+ evidence_class=evidence_class,
358
+ repeats=repeats,
359
+ verdict=verdict,
360
+ per_repeat=[row.to_row() for row in rows],
361
+ icc=icc,
362
+ within_variance=within,
363
+ divergence_step=fork_step,
364
+ determinism=determinism,
365
+ quarantined_repeats=quarantined_count,
366
+ required_env=[str(name) for name in required_env],
367
+ end_state_diff=end_state_diff,
368
+ run_id=resolved_run_id,
369
+ rung=rung,
370
+ framework=framework_name,
371
+ framework_version=framework_version,
372
+ version_requirement=version_requirement,
373
+ version_ok=version_ok_observed,
374
+ repeats_requested=repeats,
375
+ repeats_completed=len(rows),
376
+ budget_cap_s=budget,
377
+ budget_spent_s=round(budget_spent, 6),
378
+ verdict_reason=verdict_reason,
379
+ findings=findings,
380
+ artifacts_dir=str(base_dir),
381
+ )
382
+
383
+
384
+ def primary_transcript_events(result: LaneRunResult) -> list[dict[str, Any]]:
385
+ """Events of the first scoreable repeat (falling back to the first row) —
386
+ the transcript the lane normalizes into its state keys."""
387
+
388
+ from ._transcript import read_transcript
389
+
390
+ rows = [row for row in result.per_repeat if not row.get("quarantined")]
391
+ rows = rows or list(result.per_repeat)
392
+ for row in rows:
393
+ path = row.get("transcript_path")
394
+ if path and Path(str(path)).is_file():
395
+ return read_transcript(str(path))
396
+ return []
397
+
398
+
399
+ def lane_run_payload(
400
+ result: LaneRunResult,
401
+ *,
402
+ name: str | None = None,
403
+ scenario: Mapping[str, Any] | None = None,
404
+ manifest: Mapping[str, Any] | None = None,
405
+ states: Mapping[str, Any] | None = None,
406
+ metadata: Mapping[str, Any] | None = None,
407
+ ) -> dict[str, Any]:
408
+ """Serialize a lane run into the standard ``agent-learning.run.v1``
409
+ payload via the existing public envelope, with live-only fields under a
410
+ ``live_lane`` evidence block — same artifact kind, same state keys, plus
411
+ live evidence (the graduation contract, R§3.1)."""
412
+
413
+ payload: dict[str, Any] = {
414
+ "kind": AGENT_LEARNING_RUN_KIND,
415
+ "name": str(name or f"live-{result.lane}-run-{result.run_id[:8]}"),
416
+ "evidence_class": result.evidence_class,
417
+ "live_lane": result.to_block(),
418
+ "findings": list(result.findings),
419
+ "summary": {
420
+ "verdict": result.verdict,
421
+ "verdict_reason": result.verdict_reason,
422
+ "repeats": result.repeats,
423
+ "repeats_completed": result.repeats_completed,
424
+ "quarantined_repeats": result.quarantined_repeats,
425
+ "icc": result.icc,
426
+ "divergence_step": result.divergence_step,
427
+ },
428
+ }
429
+ if scenario is not None:
430
+ payload["scenario"] = dict(scenario)
431
+ if manifest is not None:
432
+ payload["manifest"] = dict(manifest)
433
+ if states:
434
+ for state_key, state_value in states.items():
435
+ payload[str(state_key)] = state_value
436
+ if metadata:
437
+ payload["metadata"] = dict(metadata)
438
+ return public_payload(payload, kind=AGENT_LEARNING_RUN_KIND)
439
+
440
+
441
+ # --- dual-channel voice evidence (3B/3C — PRD §4.2 / guide §3.5) -------------
442
+
443
+
444
+ def _activity_mask(
445
+ pcm: np.ndarray,
446
+ *,
447
+ frame_samples: int,
448
+ energy_threshold_db: float,
449
+ ) -> np.ndarray:
450
+ samples = np.asarray(pcm, dtype=float)
451
+ if samples.size == 0:
452
+ return np.zeros(0, dtype=bool)
453
+ peak = np.max(np.abs(samples))
454
+ if peak > 0:
455
+ samples = samples / peak
456
+ frame_count = int(np.ceil(samples.size / frame_samples))
457
+ padded = np.zeros(frame_count * frame_samples, dtype=float)
458
+ padded[: samples.size] = samples
459
+ frames = padded.reshape(frame_count, frame_samples)
460
+ rms = np.sqrt((frames**2).mean(axis=1))
461
+ with np.errstate(divide="ignore"):
462
+ rms_db = 20.0 * np.log10(np.where(rms > 0, rms, 1e-12))
463
+ return rms_db > energy_threshold_db
464
+
465
+
466
+ def _segments(mask: np.ndarray) -> list[tuple[int, int]]:
467
+ """Contiguous active [start, end) frame spans."""
468
+
469
+ spans: list[tuple[int, int]] = []
470
+ start: int | None = None
471
+ for index, active in enumerate(mask):
472
+ if active and start is None:
473
+ start = index
474
+ elif not active and start is not None:
475
+ spans.append((start, index))
476
+ start = None
477
+ if start is not None:
478
+ spans.append((start, len(mask)))
479
+ return spans
480
+
481
+
482
+ def derive_channel_evidence(
483
+ user_pcm: np.ndarray,
484
+ agent_pcm: np.ndarray,
485
+ *,
486
+ sample_rate: int,
487
+ frame_ms: float = 20.0,
488
+ energy_threshold_db: float = -40.0,
489
+ ) -> dict[str, Any]:
490
+ """Compute the ``channels.derived`` block from the two PCM streams —
491
+ never from transcripts (R§3.5): barge-in latency, overlap totals,
492
+ post-interrupt recovery turns, and agent onset (ttfb). Pure-numpy
493
+ energy/onset detection; rung-2+ only (rung 1 has no channels block —
494
+ the rung-1 honesty rule, guide §3.5)."""
495
+
496
+ if sample_rate <= 0:
497
+ raise ValueError("sample_rate must be positive")
498
+ frame_samples = max(int(sample_rate * frame_ms / 1000.0), 1)
499
+ user_mask = _activity_mask(
500
+ user_pcm, frame_samples=frame_samples, energy_threshold_db=energy_threshold_db
501
+ )
502
+ agent_mask = _activity_mask(
503
+ agent_pcm, frame_samples=frame_samples, energy_threshold_db=energy_threshold_db
504
+ )
505
+ width = max(len(user_mask), len(agent_mask))
506
+ user_full = np.zeros(width, dtype=bool)
507
+ agent_full = np.zeros(width, dtype=bool)
508
+ user_full[: len(user_mask)] = user_mask
509
+ agent_full[: len(agent_mask)] = agent_mask
510
+
511
+ overlap = user_full & agent_full
512
+ overlap_spans = _segments(overlap)
513
+ overlap_total_ms = float(sum(end - start for start, end in overlap_spans)) * frame_ms
514
+
515
+ user_spans = _segments(user_full)
516
+ agent_spans = _segments(agent_full)
517
+
518
+ # ttfb: first agent onset after the first user utterance ends.
519
+ ttfb_ms: float | None = None
520
+ if user_spans and agent_spans:
521
+ first_user_end = user_spans[0][1]
522
+ for start, _ in agent_spans:
523
+ if start >= first_user_end:
524
+ ttfb_ms = float(start - first_user_end) * frame_ms
525
+ break
526
+
527
+ # barge-in: first user onset that lands mid-agent-speech; latency runs
528
+ # until the agent yields (its active span ends).
529
+ barge_in_latency_ms: float | None = None
530
+ barge_frame: int | None = None
531
+ for user_start, _ in user_spans:
532
+ for agent_start, agent_end in agent_spans:
533
+ if agent_start < user_start < agent_end:
534
+ barge_in_latency_ms = float(agent_end - user_start) * frame_ms
535
+ barge_frame = user_start
536
+ break
537
+ if barge_in_latency_ms is not None:
538
+ break
539
+
540
+ # recovery: agent speech segments after the interrupt until the first
541
+ # segment that starts clear of user speech (a clean turn).
542
+ post_interrupt_recovery_turns: int | None = None
543
+ if barge_frame is not None:
544
+ turns = 0
545
+ for agent_start, _ in agent_spans:
546
+ if agent_start <= barge_frame:
547
+ continue
548
+ turns += 1
549
+ if not user_full[agent_start]:
550
+ break
551
+ post_interrupt_recovery_turns = turns
552
+
553
+ return {
554
+ "barge_in_latency_ms": barge_in_latency_ms,
555
+ "overlap_total_ms": overlap_total_ms,
556
+ "overlap_segments": len(overlap_spans),
557
+ "post_interrupt_recovery_turns": post_interrupt_recovery_turns,
558
+ "ttfb_ms": ttfb_ms,
559
+ "frame_ms": frame_ms,
560
+ "energy_threshold_db": energy_threshold_db,
561
+ }