agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/live/_codec.py ADDED
@@ -0,0 +1,391 @@
1
+ """Phase 9A unit 3 — pure-numpy codec-survival stage (the Phase-12-reserved home).
2
+
3
+ ARCH §2.2 / decisions 9A-A2 (G.711 μ-law/A-law PURE-NUMPY v1, ZERO new dep;
4
+ Opus-NB/AMR a post-v1 build-dep extra, auto-skip), 9A-A3 (module home), 9A-A7
5
+ (no neural codec; registry-extensible), 9A-A11 (default-ON at rung-2), 9A-A12
6
+ (facade ``score_codec_survival`` / ``CodecUnsupportedError``), 9A-A13 (computed
7
+ ``phone_survival`` 4 frozen + 3 computed fields), 9A-D4.
8
+
9
+ Imports: numpy + STDLIB ONLY. G.711 μ-law/A-law are vectorized numpy companding
10
+ tables — NOT ``audioop`` (deprecated 3.11, REMOVED 3.13 per PEP 594; the kit's
11
+ dev interpreter is 3.14 → ``audioop`` cannot back it). 8 kHz resample is
12
+ pure-numpy decimation (no scipy). Gilbert-Elliott packet loss is seeded numpy.
13
+ The codec/packet operators raise on text-rung input exactly as the ``_perturb``
14
+ acoustic operators do (a codec round-trip on a transcript is a contract error).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from typing import Any
20
+
21
+ import numpy as np
22
+
23
+ # --- closed-vocabulary constants (mirrored in trinity.py, cross-pinned by a
24
+ # unit test — the GUNA_AXES cross-pin pattern; trinity is the gate-pinned home).
25
+ V1_VOICE_CODECS = ("g711_ulaw", "g711_alaw", "opus_nb", "amr_nb")
26
+ # g711_* = v1 pure-numpy; opus_nb/amr_nb = post-v1 build-dep, auto-skip.
27
+ _V1_NUMPY_CODECS = ("g711_ulaw", "g711_alaw")
28
+ _POST_V1_CODECS = ("opus_nb", "amr_nb")
29
+ V1_VOICE_PACKET_LOSS_MODELS = ("gilbert_elliott",)
30
+ V1_VOICE_CODEC_PROFILES = (
31
+ "g711_ulaw_8k_ge",
32
+ "g711_alaw_8k_ge",
33
+ "opus_nb_8k_ge",
34
+ "amr_nb_8k_ge",
35
+ "none",
36
+ )
37
+ # named bundle → (codec, packet_loss_model)
38
+ _PROFILE_BUNDLE = {
39
+ "g711_ulaw_8k_ge": ("g711_ulaw", "gilbert_elliott"),
40
+ "g711_alaw_8k_ge": ("g711_alaw", "gilbert_elliott"),
41
+ "opus_nb_8k_ge": ("opus_nb", "gilbert_elliott"),
42
+ "amr_nb_8k_ge": ("amr_nb", "gilbert_elliott"),
43
+ }
44
+ _TELEPHONY_RATE = 8000
45
+
46
+ # the post-v1 install path named in the auto-skip refusal (9A-A2).
47
+ _POST_V1_EXTRA = "agent-learning-kit[voice-codecs]"
48
+
49
+
50
+ class CodecUnsupportedError(RuntimeError):
51
+ """Raised when a requested codec is in ``V1_VOICE_CODECS`` but its build-dep
52
+ extra is absent (the post-v1 ``opus_nb``/``amr_nb`` path). Callers can
53
+ ``except CodecUnsupportedError`` to auto-skip exactly as the framework lanes
54
+ skip on a missing extra (the ``LANE_EXTRAS`` discipline). G.711 / packet-loss
55
+ never raise (numpy, always available)."""
56
+
57
+ def __init__(self, message: str, *, codec: str, install: str) -> None:
58
+ super().__init__(message)
59
+ self.codec = codec
60
+ self.install = install
61
+
62
+
63
+ def _require_pcm(pcm: Any, *, where: str) -> np.ndarray:
64
+ """Type-guard the input as numpy PCM; a text/str input raises a ValueError
65
+ mirroring ``_perturb.py``'s rung-wall message (the §3.4 generalization)."""
66
+
67
+ if isinstance(pcm, (str, bytes)):
68
+ raise ValueError(
69
+ f"{where} needs a real audio channel (rung 2 loopback transport or "
70
+ "above); a text/transcript input is a contract error"
71
+ )
72
+ arr = np.asarray(pcm, dtype=np.float32)
73
+ if arr.ndim != 1:
74
+ arr = arr.reshape(-1)
75
+ return arr
76
+
77
+
78
+ def resample_8k(pcm: np.ndarray, *, source_rate: int) -> np.ndarray:
79
+ """24 kHz → 8 kHz telephony band-limit via pure-numpy anti-alias + decimation
80
+ (NO scipy). Anti-alias by a simple moving-average low-pass at the target
81
+ Nyquist (4 kHz), then decimate to ``target_rate=8000``. Deterministic."""
82
+
83
+ samples = _require_pcm(pcm, where="resample_8k")
84
+ if source_rate <= 0:
85
+ raise ValueError("source_rate must be positive")
86
+ if samples.size == 0 or source_rate == _TELEPHONY_RATE:
87
+ return samples
88
+ factor = source_rate / float(_TELEPHONY_RATE)
89
+ if factor <= 1.0:
90
+ # upsampling is out of scope for the telephony band-limit; pass through.
91
+ return samples
92
+ # anti-alias: moving average window ~ decimation factor (cuts > 4 kHz energy)
93
+ window = max(int(round(factor)), 1)
94
+ if window > 1:
95
+ kernel = np.ones(window, dtype=np.float32) / float(window)
96
+ samples = np.convolve(samples, kernel, mode="same").astype(np.float32)
97
+ n_out = int(samples.size / factor)
98
+ if n_out <= 0:
99
+ return np.zeros(0, dtype=np.float32)
100
+ idx = (np.arange(n_out, dtype=np.float64) * factor).astype(np.int64)
101
+ idx = np.clip(idx, 0, samples.size - 1)
102
+ return samples[idx].astype(np.float32, copy=False)
103
+
104
+
105
+ # --- G.711 μ-law companding (vectorized numpy, ITU-T G.711) -----------------
106
+ _MU = 255.0
107
+ _ALAW_A = 87.6
108
+
109
+
110
+ def g711_ulaw_roundtrip(pcm: np.ndarray) -> np.ndarray:
111
+ """μ-law companding round-trip via vectorized numpy (NOT audioop): encode
112
+ (linear → μ-law 8-bit code) then decode (μ-law code → linear). Lossy and
113
+ deterministic — the v1 telephony codec."""
114
+
115
+ x = _require_pcm(pcm, where="g711_ulaw_roundtrip")
116
+ if x.size == 0:
117
+ return x
118
+ x = np.clip(x, -1.0, 1.0)
119
+ # encode: μ-law compression → quantize to 8-bit code
120
+ sign = np.sign(x)
121
+ magnitude = np.log1p(_MU * np.abs(x)) / np.log1p(_MU)
122
+ code = np.round(magnitude * 127.0).astype(np.int32) # 8-bit magnitude quant
123
+ code = np.clip(code, 0, 127)
124
+ # decode: μ-law expansion of the quantized code
125
+ mag_q = code.astype(np.float32) / 127.0
126
+ decoded = sign * (np.expm1(mag_q * np.log1p(_MU)) / _MU)
127
+ return decoded.astype(np.float32, copy=False)
128
+
129
+
130
+ def g711_alaw_roundtrip(pcm: np.ndarray) -> np.ndarray:
131
+ """A-law companding round-trip, same shape as the μ-law path."""
132
+
133
+ x = _require_pcm(pcm, where="g711_alaw_roundtrip")
134
+ if x.size == 0:
135
+ return x
136
+ x = np.clip(x, -1.0, 1.0)
137
+ sign = np.sign(x)
138
+ ax = np.abs(x)
139
+ ln_a = 1.0 + np.log(_ALAW_A)
140
+ low = ax < (1.0 / _ALAW_A)
141
+ compressed = np.where(
142
+ low,
143
+ (_ALAW_A * ax) / ln_a,
144
+ (1.0 + np.log(np.clip(_ALAW_A * ax, 1e-12, None))) / ln_a,
145
+ )
146
+ code = np.clip(np.round(compressed * 127.0).astype(np.int32), 0, 127)
147
+ comp_q = code.astype(np.float32) / 127.0
148
+ # A-law expansion (inverse)
149
+ thresh = 1.0 / ln_a
150
+ decoded_mag = np.where(
151
+ comp_q < thresh,
152
+ (comp_q * ln_a) / _ALAW_A,
153
+ np.exp(comp_q * ln_a - 1.0) / _ALAW_A,
154
+ )
155
+ decoded = sign * decoded_mag
156
+ return decoded.astype(np.float32, copy=False)
157
+
158
+
159
+ def gilbert_elliott_loss(
160
+ pcm: np.ndarray,
161
+ *,
162
+ loss_avg: float = 0.02,
163
+ burst_ms: float = 100.0,
164
+ sample_rate: int = _TELEPHONY_RATE,
165
+ seed: int,
166
+ ) -> tuple[np.ndarray, dict]:
167
+ """Two-state burst packet loss (the τ-Voice default 2 %/100 ms recipe,
168
+ R§2.2). Pure-numpy, seeded via ``np.random.default_rng(seed)`` →
169
+ reproducible under seed. Returns the degraded PCM + a record
170
+ ``{model, loss_avg, burst_ms, seed, loss_realized}`` for the artifact/replay.
171
+ """
172
+
173
+ x = _require_pcm(pcm, where="gilbert_elliott_loss")
174
+ if x.size == 0 or loss_avg <= 0:
175
+ return x, {
176
+ "model": "gilbert_elliott",
177
+ "loss_avg": float(loss_avg),
178
+ "burst_ms": float(burst_ms),
179
+ "seed": int(seed),
180
+ "loss_realized": 0.0,
181
+ }
182
+ rng = np.random.default_rng(seed)
183
+ frame_samples = max(int(sample_rate * 20.0 / 1000.0), 1) # 20 ms frames
184
+ n_frames = max(int(np.ceil(x.size / frame_samples)), 1)
185
+ # two-state Markov chain: G(ood) and B(ad). Mean burst length = burst_ms/20ms
186
+ # frames ⇒ p(B→G) = 1/burst_frames; steady-state loss = loss_avg ⇒ derive
187
+ # p(G→B) from the balance equation π_B = loss_avg.
188
+ burst_frames = max(burst_ms / 20.0, 1.0)
189
+ p_bg = 1.0 / burst_frames # leave Bad
190
+ # π_B = p_gb / (p_gb + p_bg) = loss_avg ⇒ p_gb = loss_avg * p_bg / (1-loss_avg)
191
+ p_gb = (loss_avg * p_bg) / max(1.0 - loss_avg, 1e-9)
192
+ degraded = x.copy()
193
+ bad = False
194
+ lost_frames = 0
195
+ for f in range(n_frames):
196
+ if bad:
197
+ if rng.random() < p_bg:
198
+ bad = False
199
+ else:
200
+ if rng.random() < p_gb:
201
+ bad = True
202
+ if bad:
203
+ start = f * frame_samples
204
+ degraded[start : start + frame_samples] = 0.0
205
+ lost_frames += 1
206
+ record = {
207
+ "model": "gilbert_elliott",
208
+ "loss_avg": float(loss_avg),
209
+ "burst_ms": float(burst_ms),
210
+ "seed": int(seed),
211
+ "loss_realized": round(lost_frames / float(n_frames), 6),
212
+ }
213
+ return degraded.astype(np.float32, copy=False), record
214
+
215
+
216
+ def _codec_roundtrip(pcm: np.ndarray, *, codec: str) -> np.ndarray:
217
+ if codec == "g711_ulaw":
218
+ return g711_ulaw_roundtrip(pcm)
219
+ if codec == "g711_alaw":
220
+ return g711_alaw_roundtrip(pcm)
221
+ if codec in _POST_V1_CODECS:
222
+ raise CodecUnsupportedError(
223
+ f"codec {codec!r} is a post-v1 build-dep extra and is not installed; "
224
+ f"install {_POST_V1_EXTRA} to enable it (v1 ships g711_ulaw/g711_alaw "
225
+ "as the required pure-numpy codecs)",
226
+ codec=codec,
227
+ install=_POST_V1_EXTRA,
228
+ )
229
+ raise ValueError(f"unknown codec {codec!r}; expected one of {V1_VOICE_CODECS}")
230
+
231
+
232
+ def _apply_channel(
233
+ pcm: np.ndarray, *, codec: str, packet_loss: str, seed: int, sample_rate: int
234
+ ) -> tuple[np.ndarray, dict]:
235
+ """Resample → codec round-trip → packet loss, returning degraded PCM + the
236
+ codec_round_trip record (UI-UX §3.3 shape)."""
237
+
238
+ if packet_loss not in V1_VOICE_PACKET_LOSS_MODELS:
239
+ raise ValueError(
240
+ f"packet_loss_model {packet_loss!r} must be one of "
241
+ f"{V1_VOICE_PACKET_LOSS_MODELS}"
242
+ )
243
+ resampled = resample_8k(pcm, source_rate=sample_rate)
244
+ coded = _codec_roundtrip(resampled, codec=codec)
245
+ degraded, loss_record = gilbert_elliott_loss(
246
+ coded, sample_rate=_TELEPHONY_RATE, seed=seed
247
+ )
248
+ record = {
249
+ "codec": codec,
250
+ "resampled_to_hz": _TELEPHONY_RATE,
251
+ "source_rate_hz": int(sample_rate),
252
+ "packet_loss": loss_record,
253
+ "seed": int(seed),
254
+ }
255
+ return degraded, record
256
+
257
+
258
+ def apply_codec_profile(
259
+ user_pcm: np.ndarray,
260
+ agent_pcm: np.ndarray,
261
+ *,
262
+ profile: str,
263
+ seed: int,
264
+ sample_rate: int,
265
+ ) -> tuple[np.ndarray, np.ndarray, dict]:
266
+ """Apply a named codec profile (codec + resample + packet-loss bundle) to
267
+ both streams. ``profile='none'`` is a no-op (clean-PCM loopback). Raises
268
+ ``CodecUnsupportedError`` for ``opus_nb_8k_ge``/``amr_nb_8k_ge`` when the
269
+ extra is absent (post-v1). Returns degraded (user_pcm, agent_pcm) + the
270
+ codec_round_trip record (UI-UX §3.3 shape)."""
271
+
272
+ if profile not in V1_VOICE_CODEC_PROFILES:
273
+ raise ValueError(
274
+ f"codec_profile {profile!r} must be one of {V1_VOICE_CODEC_PROFILES}"
275
+ )
276
+ if profile == "none":
277
+ return (
278
+ _require_pcm(user_pcm, where="apply_codec_profile"),
279
+ _require_pcm(agent_pcm, where="apply_codec_profile"),
280
+ {"profile": "none", "applied": False},
281
+ )
282
+ codec, packet_loss = _PROFILE_BUNDLE[profile]
283
+ user_deg, user_rec = _apply_channel(
284
+ user_pcm, codec=codec, packet_loss=packet_loss, seed=seed, sample_rate=sample_rate
285
+ )
286
+ agent_deg, agent_rec = _apply_channel(
287
+ agent_pcm,
288
+ codec=codec,
289
+ packet_loss=packet_loss,
290
+ seed=seed + 1,
291
+ sample_rate=sample_rate,
292
+ )
293
+ record = {
294
+ "profile": profile,
295
+ "applied": True,
296
+ "codec": codec,
297
+ "packet_loss_model": packet_loss,
298
+ "resampled_to_hz": _TELEPHONY_RATE,
299
+ "source_rate_hz": int(sample_rate),
300
+ "user": user_rec,
301
+ "agent": agent_rec,
302
+ "seed": int(seed),
303
+ }
304
+ return user_deg, agent_deg, record
305
+
306
+
307
+ def _band_energy_lt_4khz(pcm: np.ndarray, *, sample_rate: int) -> float:
308
+ """Fraction of signal energy below 4 kHz the telephony codec preserves
309
+ (CodecAttack <4 kHz framing, R§2.2)."""
310
+
311
+ x = _require_pcm(pcm, where="band_energy")
312
+ if x.size < 2:
313
+ return 0.0
314
+ spectrum = np.abs(np.fft.rfft(x)) ** 2
315
+ freqs = np.fft.rfftfreq(x.size, d=1.0 / float(sample_rate))
316
+ total = float(spectrum.sum())
317
+ if total <= 0:
318
+ return 0.0
319
+ low = float(spectrum[freqs < 4000.0].sum())
320
+ return round(low / total, 6)
321
+
322
+
323
+ def _success_score(pcm: np.ndarray) -> float:
324
+ """A reproducible proxy success score for the channel pre/post twins: RMS
325
+ energy normalized to [0,1]. The clean twin scores higher than the degraded
326
+ twin (the channel attenuates / drops frames), so the pre→post delta is the
327
+ survival evidence. Deterministic — no model call."""
328
+
329
+ x = _require_pcm(pcm, where="success_score")
330
+ if x.size == 0:
331
+ return 0.0
332
+ rms = float(np.sqrt((x.astype(np.float64) ** 2).mean()))
333
+ return round(min(rms * 2.0, 1.0), 6)
334
+
335
+
336
+ def score_codec_survival(
337
+ user_pcm: np.ndarray,
338
+ agent_pcm: np.ndarray,
339
+ *,
340
+ codec: str,
341
+ packet_loss: str,
342
+ seed: int,
343
+ sample_rate: int = 24000,
344
+ pre_channel_success: float | None = None,
345
+ ) -> dict:
346
+ """Re-validate an acoustic claim through the telephony channel; return the
347
+ COMPUTED ``phone_survival`` object (9A-A13). The Phase-12 frozen 4 fields
348
+ (``status``/``tier``/``scope_label?``/``reason``) keep their vocabulary
349
+ unchanged; 3 OPTIONAL computed-evidence fields
350
+ (``pre_channel_success``/``post_channel_success``/``band_energy_lt_4khz``)
351
+ are present ONLY when ``tier ∈ {channel_simulated, channel_live}``."""
352
+
353
+ _require_pcm(user_pcm, where="score_codec_survival") # type-guard the user side
354
+ agent = _require_pcm(agent_pcm, where="score_codec_survival")
355
+ # the clean twin success (BEFORE the channel) — measured on the agent side
356
+ # (the side carrying the claim under test) unless supplied by the caller.
357
+ pre = (
358
+ float(pre_channel_success)
359
+ if pre_channel_success is not None
360
+ else _success_score(agent)
361
+ )
362
+ agent_deg, channel_record = _apply_channel(
363
+ agent, codec=codec, packet_loss=packet_loss, seed=seed, sample_rate=sample_rate
364
+ )
365
+ post = _success_score(agent_deg)
366
+ band = _band_energy_lt_4khz(agent_deg, sample_rate=_TELEPHONY_RATE)
367
+
368
+ # status derives from the pre→post delta
369
+ if pre <= 0:
370
+ status = "untested"
371
+ else:
372
+ retained = post / pre if pre > 0 else 0.0
373
+ if retained >= 0.85:
374
+ status = "survives"
375
+ elif retained >= 0.4:
376
+ status = "partial"
377
+ else:
378
+ status = "dies"
379
+
380
+ return {
381
+ "status": status,
382
+ "tier": "channel_simulated",
383
+ "reason": (
384
+ f"codec={codec} packet_loss={packet_loss} "
385
+ f"loss_realized={channel_record['packet_loss']['loss_realized']} "
386
+ f"pre={pre} post={post} (8 kHz telephony channel, simulated)"
387
+ ),
388
+ "pre_channel_success": round(pre, 6),
389
+ "post_channel_success": round(post, 6),
390
+ "band_energy_lt_4khz": band,
391
+ }
@@ -0,0 +1,134 @@
1
+ """Live-lane contract: vocabularies, lane specs, budgets, flag discipline.
2
+
3
+ Imports: stdlib only. Every substrate module must remain importable (and
4
+ its unit tests green) in an environment with no framework extra installed —
5
+ the live_lane_boundary gate scans them like any other release module.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import dataclasses
11
+ import os
12
+ from typing import Any, Mapping
13
+
14
+ # --- artifact kind (same kind simulate/cli emit — never a parallel kind) ----
15
+ AGENT_LEARNING_RUN_KIND = "agent-learning.run.v1"
16
+
17
+ # --- evidence classes (R§3.2; PRD §4.1) -----------------------------------
18
+ EVIDENCE_CLASSES = ("local_gate", "live_lane", "live_stressed", "captured_fixture")
19
+ RELEASE_ADMISSIBLE_EVIDENCE_CLASSES = ("local_gate", "captured_fixture")
20
+
21
+ # --- failure layers (R§1 #1 HarnessFix; PRD §4.1) --------------------------
22
+ FAILURE_LAYERS = ("lane_infra", "framework_runtime", "provider", "agent_behavior")
23
+
24
+ # --- per-scenario verdicts (R§3.4) ------------------------------------------
25
+ VERDICTS = ("pass", "fail", "unstable", "void")
26
+
27
+ # --- env-flag conventions (PRD §4.1: AGENT_LEARNING_LIVE_<LANE>=1) ---------
28
+ LANE_ENV_FLAGS = {
29
+ "livekit": "AGENT_LEARNING_LIVE_LIVEKIT",
30
+ "pipecat": "AGENT_LEARNING_LIVE_PIPECAT",
31
+ "langchain": "AGENT_LEARNING_LIVE_LANGCHAIN",
32
+ "mcp": "AGENT_LEARNING_LIVE_MCP",
33
+ "a2a": "AGENT_LEARNING_LIVE_A2A",
34
+ "credentialed": "AGENT_LEARNING_LIVE_CREDENTIALED",
35
+ }
36
+
37
+ # --- lane → extra map (skip lines and import errors name these) ------------
38
+ LANE_EXTRAS = {
39
+ "livekit": "livekit",
40
+ "pipecat": "pipecat",
41
+ "langchain": "langchain",
42
+ "mcp": "mcp",
43
+ "a2a": "a2a",
44
+ }
45
+
46
+ # --- budget caps (P3-D2): 600 s default; voice lanes 900 s -----------------
47
+ LANE_BUDGET_S_DEFAULT = 600.0
48
+ LANE_BUDGET_S = {"livekit": 900.0, "pipecat": 900.0}
49
+
50
+ # --- repeat policy (P3-D2) ---------------------------------------------------
51
+ DEFAULT_REPEATS = 8
52
+ UNSTABLE_ICC_FLOOR = 0.5
53
+
54
+
55
+ def lane_budget_s(lane: str) -> float:
56
+ """Hard wall-clock cap for one lane run (P3-D2)."""
57
+
58
+ return LANE_BUDGET_S.get(lane, LANE_BUDGET_S_DEFAULT)
59
+
60
+
61
+ class LaneDisabledError(RuntimeError):
62
+ """Raised when a lane entry point runs without its env flag."""
63
+
64
+
65
+ def require_lane_enabled(lane: str) -> None:
66
+ """Gate every lane entry on its env flag (PRD §4.1).
67
+
68
+ The live_lane_boundary gate statically asserts every public lane module
69
+ calls this (unit 4.2 check 3) — the dynamic raise and the static scan
70
+ are two halves of the same discipline.
71
+ """
72
+
73
+ flag = LANE_ENV_FLAGS[lane]
74
+ if os.environ.get(flag) != "1":
75
+ raise LaneDisabledError(
76
+ f"live lane '{lane}' is opt-in: set {flag}=1 to run it "
77
+ "(never set in release flows)"
78
+ )
79
+
80
+
81
+ @dataclasses.dataclass(frozen=True)
82
+ class LaneSpec:
83
+ """What a lane run was asked to do — shared by runner and lanes."""
84
+
85
+ lane: str
86
+ scenario: Mapping[str, Any]
87
+ rung: int = 1
88
+ required_env: tuple[str, ...] = ()
89
+ version_requirement: str | None = None
90
+ repeats: int = DEFAULT_REPEATS
91
+ budget_s: float | None = None
92
+ evidence_class: str = "live_lane"
93
+
94
+ def resolved_budget_s(self) -> float:
95
+ if self.budget_s is not None:
96
+ return float(self.budget_s)
97
+ return lane_budget_s(self.lane)
98
+
99
+
100
+ @dataclasses.dataclass
101
+ class LaneRun:
102
+ """One repeat of one scenario inside a lane run (a per_repeat row)."""
103
+
104
+ index: int
105
+ passed: bool | None
106
+ score: float | None
107
+ failure_layer: str | None
108
+ quarantined: bool
109
+ evidence_class: str
110
+ detail: str = ""
111
+ void_reason: str | None = None
112
+ transcript_path: str | None = None
113
+ transcript_complete: bool | None = None
114
+ transcript_sha256: str | None = None
115
+ step_signature: tuple[str, ...] = ()
116
+
117
+ def to_row(self) -> dict[str, Any]:
118
+ row: dict[str, Any] = {
119
+ "repeat": self.index,
120
+ "passed": self.passed,
121
+ "score": self.score,
122
+ "failure_layer": self.failure_layer,
123
+ "quarantined": self.quarantined,
124
+ "evidence_class": self.evidence_class,
125
+ "transcript_path": self.transcript_path,
126
+ "transcript_complete": self.transcript_complete,
127
+ "transcript_sha256": self.transcript_sha256,
128
+ "step_signature": list(self.step_signature),
129
+ }
130
+ if self.detail:
131
+ row["detail"] = self.detail
132
+ if self.void_reason:
133
+ row["void_reason"] = self.void_reason
134
+ return row