agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,376 @@
1
+ """LiveKit live lane (3B) — real ``livekit-agents`` AgentSession, opt-in.
2
+
3
+ Framework imports: NONE at module top (P3-D1). Rung-1 execution happens in
4
+ the ``_workers/livekit_worker.py`` subprocess (the only sanctioned top-level
5
+ framework import home); this module is importable in the no-extras release
6
+ env and the live_lane_boundary gate scans it like any release module.
7
+
8
+ Rungs (P3-D3): 1 virtual-clock text driver (default, implemented) →
9
+ 2 loopback real-transport audio → 3 LiveKit Cloud/SIP (``live_credentialed``,
10
+ standard LiveKit credential names). Rung 1 is honest about its tier: timing-only voice metrics,
11
+ no ``channels`` block, no audio claims (guide §3.5).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import tempfile
17
+ import uuid
18
+ from pathlib import Path
19
+ from typing import Any, Mapping, Optional, Sequence
20
+
21
+ from ._contract import lane_budget_s, require_lane_enabled
22
+ from ._perturb import apply_text_perturbations, perturbations_stanza
23
+ from ._runner import run_worker_once
24
+ from ._stats import (
25
+ derive_channel_evidence,
26
+ lane_run_payload,
27
+ primary_transcript_events,
28
+ run_repeated,
29
+ )
30
+
31
+ _WORKERS = Path(__file__).resolve().parent / "_workers"
32
+ _RUNG_LABELS = {1: "virtual_clock", 2: "loopback_transport", 3: "cloud_sip"}
33
+
34
+ # Rung-3 credential names: exactly the names the vendored engine reads
35
+ # (engines/livekit.py reads LIVEKIT_API_KEY/LIVEKIT_API_SECRET; the server
36
+ # URL arrives via LIVEKIT_URL, P3-D5).
37
+ RUNG3_REQUIRED_ENV = ("LIVEKIT_URL", "LIVEKIT_API_KEY", "LIVEKIT_API_SECRET")
38
+
39
+ _DEFAULT_TURNS = (
40
+ {"user": "Hello, can you hear me?"},
41
+ {"user": "Great - please confirm my appointment for tomorrow."},
42
+ )
43
+
44
+
45
+ def _scenario_turns(scenario: Mapping[str, Any]) -> list[dict[str, Any]]:
46
+ raw = scenario.get("turns") or scenario.get("user_messages")
47
+ if not raw:
48
+ return [dict(turn) for turn in _DEFAULT_TURNS]
49
+ turns: list[dict[str, Any]] = []
50
+ for item in raw:
51
+ if isinstance(item, str):
52
+ turns.append({"user": item})
53
+ elif isinstance(item, Mapping):
54
+ turns.append(dict(item))
55
+ return turns or [dict(turn) for turn in _DEFAULT_TURNS]
56
+
57
+
58
+ def _voice_timing(events: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
59
+ """Timing-only voice metrics (the rung-1 honesty tier): per-turn agent
60
+ response latency derived from event timestamps — no audio claims."""
61
+
62
+ latencies_ms: list[float] = []
63
+ pending_user_t: float | None = None
64
+ for event in events:
65
+ channel = event.get("channel")
66
+ if channel == "user" and event.get("type") == "message":
67
+ t = event.get("t")
68
+ pending_user_t = float(t) if isinstance(t, (int, float)) else None
69
+ elif channel == "agent" and event.get("type") == "message":
70
+ t = event.get("t")
71
+ if pending_user_t is not None and isinstance(t, (int, float)):
72
+ latencies_ms.append(round((float(t) - pending_user_t) * 1000.0, 3))
73
+ pending_user_t = None
74
+ return {
75
+ "turn_latencies_ms": latencies_ms,
76
+ "mean_turn_latency_ms": (
77
+ round(sum(latencies_ms) / len(latencies_ms), 3) if latencies_ms else None
78
+ ),
79
+ }
80
+
81
+
82
+ def _realtime_state(
83
+ events: Sequence[Mapping[str, Any]], *, rung_label: str
84
+ ) -> dict[str, Any]:
85
+ items = []
86
+ for index, event in enumerate(events, start=1):
87
+ if event.get("channel") in ("user", "agent", "tool"):
88
+ payload = event.get("payload")
89
+ payload = payload if isinstance(payload, Mapping) else {}
90
+ items.append(
91
+ {
92
+ "index": index,
93
+ "channel": event.get("channel"),
94
+ "item_type": event.get("type"),
95
+ "text": payload.get("text"),
96
+ }
97
+ )
98
+ return {
99
+ "engine": "live_lane_livekit",
100
+ "rung": rung_label,
101
+ "item_count": len(items),
102
+ "items": items[:200],
103
+ }
104
+
105
+
106
+ def _rung2_loopback_channels(
107
+ turns: Sequence[Mapping[str, Any]],
108
+ *,
109
+ loopback: Optional[Mapping[str, Any]],
110
+ codec_profile: str,
111
+ seed: int,
112
+ acoustic_operators: Sequence[str] = (),
113
+ ) -> tuple[dict[str, Any], str, list[dict[str, Any]]]:
114
+ """Phase 9A unit 2 + Phase-12 12C rung-2 — the rung-2 loopback dispatch
115
+ (§2.1 / §2.5 + ARCH §2c).
116
+
117
+ Produce the two PCM streams via the deterministic ``_loopback`` round-trip,
118
+ apply the rung-2 ACOUSTIC operators (Phase-12 12C: ``mix_noise`` /
119
+ ``mix_interference`` / ``reverb_blend`` over the user PCM — the attack the
120
+ framework hears) BEFORE the codec stage, apply the default-ON codec
121
+ round-trip (9A-A11) unless ``codec_profile == "none"``, feed the
122
+ ALREADY-BUILT ``derive_channel_evidence`` (REUSED, NOT rebuilt), and return
123
+ the ``channels`` block + the ``fidelity_tier`` marker + the applied acoustic
124
+ operator records (the paired-clean stanza). The loopback module is reached
125
+ via the sanctioned ``from fi.alk import live`` function-body idiom so
126
+ this module stays framework-free and the ``live_lane_boundary`` import
127
+ discipline holds.
128
+
129
+ The codec-survival score is computed on the PERTURBED-then-channel signal so
130
+ ``phone_survival`` honestly reflects whether the acoustic attack reproduces
131
+ through the 8 kHz telephony channel (P12-D2): no ``survives``/``partial``
132
+ claim without a codec record."""
133
+
134
+ from fi.alk import live # sanctioned facade idiom (cli.py)
135
+
136
+ cfg = dict(loopback or {})
137
+ tick_ms = float(cfg.get("tick_ms", live._loopback.DEFAULT_TICK_MS))
138
+ sample_rate = int(cfg.get("sample_rate", live._loopback.DEFAULT_SAMPLE_RATE))
139
+ loop_seed = int(cfg.get("seed", seed))
140
+ profile = str(cfg.get("codec_profile", codec_profile))
141
+
142
+ loop = live._loopback.run_loopback_roundtrip(
143
+ list(turns),
144
+ user_wav=cfg.get("user_wav"),
145
+ agent_wav=cfg.get("agent_wav"),
146
+ tick_ms=tick_ms,
147
+ sample_rate=sample_rate,
148
+ seed=loop_seed,
149
+ )
150
+ user_pcm, agent_pcm = loop["user_pcm"], loop["agent_pcm"]
151
+
152
+ # Phase-12 12C rung-2: the acoustic attack rides the USER channel (the side
153
+ # the framework hears). Applied to the CLEAN loopback PCM before the codec
154
+ # stage; deterministic under loop_seed. The agent side is untouched.
155
+ acoustic_applied: list[dict[str, Any]] = []
156
+ attacked_user_pcm = user_pcm
157
+ if acoustic_operators:
158
+ attacked_user_pcm, acoustic_applied = live._perturb.apply_acoustic_perturbations(
159
+ user_pcm,
160
+ list(acoustic_operators),
161
+ seed=loop_seed,
162
+ sample_rate=sample_rate,
163
+ )
164
+ user_pcm = attacked_user_pcm
165
+
166
+ codec_record: dict[str, Any] | None = None
167
+ phone_survival: dict[str, Any] | None = None
168
+ if profile != "none":
169
+ user_pcm, agent_pcm, codec_record = live._codec.apply_codec_profile(
170
+ user_pcm, agent_pcm, profile=profile, seed=loop_seed, sample_rate=sample_rate
171
+ )
172
+ codec, packet_loss = live._codec._PROFILE_BUNDLE[profile]
173
+ # the attack rides the USER channel, so re-validate the user side through
174
+ # the channel (the clean user PCM is the pre-channel twin).
175
+ phone_survival = live._codec.score_codec_survival(
176
+ loop["user_pcm"],
177
+ attacked_user_pcm,
178
+ codec=codec,
179
+ packet_loss=packet_loss,
180
+ seed=loop_seed,
181
+ sample_rate=sample_rate,
182
+ )
183
+
184
+ derived = derive_channel_evidence(
185
+ user_pcm, agent_pcm, sample_rate=(8000 if profile != "none" else sample_rate)
186
+ )
187
+ channels: dict[str, Any] = {
188
+ "derived": derived,
189
+ "source": "derive_channel_evidence",
190
+ "rung": _RUNG_LABELS[2],
191
+ "fidelity_tier": "deterministic_loopback",
192
+ "seed": loop_seed,
193
+ "loopback_provenance": loop["provenance"],
194
+ }
195
+ if codec_record is not None:
196
+ channels["codec_round_trip"] = codec_record
197
+ if phone_survival is not None:
198
+ channels["phone_survival"] = phone_survival
199
+ if acoustic_applied:
200
+ channels["acoustic_operators"] = acoustic_applied
201
+ return channels, "deterministic_loopback", acoustic_applied
202
+
203
+
204
+ def run_livekit_lane(
205
+ scenario: Mapping[str, Any],
206
+ *,
207
+ rung: int = 1, # P3-D3: 1 virtual-clock | 2 loopback transport | 3 cloud/SIP
208
+ repeats: int = 8,
209
+ stressed: bool = False, # perturbation sub-lane -> evidence_class "live_stressed"
210
+ perturbations: Optional[Sequence[str]] = None,
211
+ seed: int = 0,
212
+ required_env: Optional[Sequence[str]] = None,
213
+ version_requirement: str | None = None,
214
+ budget_s: float | None = None,
215
+ artifacts_dir: str | Path | None = None,
216
+ # Phase 9A (BBG A2): additive optional loopback config consumed ONLY on the
217
+ # rung==2 branch; rung-1/rung-3 callers are unaffected.
218
+ loopback: Optional[Mapping[str, Any]] = None,
219
+ codec_profile: str = "g711_ulaw_8k_ge",
220
+ ) -> dict[str, Any]:
221
+ require_lane_enabled("livekit")
222
+ if rung >= 3:
223
+ require_lane_enabled("credentialed")
224
+ if rung not in _RUNG_LABELS:
225
+ raise ValueError(f"rung must be one of {sorted(_RUNG_LABELS)}, got {rung}")
226
+
227
+ required = tuple(required_env) if required_env is not None else ()
228
+ operators = list(perturbations or (["asr_error"] if stressed else []))
229
+ turns = _scenario_turns(scenario)
230
+ # Phase-12 12C rung-2: split text-rung operators (applied to the turn script)
231
+ # from acoustic operators (applied to the rung-2 loopback PCM). At rung-1 an
232
+ # acoustic operator still raises inside ``apply_text_perturbations`` (the
233
+ # rung wall is unchanged for text-rung input).
234
+ from ._perturb import ACOUSTIC_RUNG_OPERATORS
235
+
236
+ acoustic_operators = [op for op in operators if op in ACOUSTIC_RUNG_OPERATORS]
237
+ text_operators = [op for op in operators if op not in ACOUSTIC_RUNG_OPERATORS]
238
+ if rung != 2 and acoustic_operators:
239
+ # acoustic operators require the rung-2 PCM channel; outside it they hit
240
+ # the same rung wall ``apply_text_perturbations`` enforces (no silent
241
+ # acoustic claim before the audio channel exists — ARCH §2c).
242
+ raise ValueError(
243
+ f"acoustic operators {acoustic_operators} need a real audio channel "
244
+ "(rung 2 loopback transport or above); rung "
245
+ f"{rung} ({_RUNG_LABELS[rung]}) is a text-rung tier"
246
+ )
247
+ applied: list[dict[str, Any]] = []
248
+ if text_operators:
249
+ turns, applied = apply_text_perturbations(turns, text_operators, seed=seed)
250
+
251
+ # Phase 9A unit 2: the rung wall narrows — rung-2 dispatches into the
252
+ # deterministic loopback (§2.1); rung-3 still raises (the owner live-proof,
253
+ # unit 7). rung-1 is completely untouched (timing-only, NO channels block).
254
+ channels: dict[str, Any] | None = None
255
+ fidelity_tier: str | None = None
256
+ acoustic_applied: list[dict[str, Any]] = []
257
+ if rung == 2:
258
+ channels, fidelity_tier, acoustic_applied = _rung2_loopback_channels(
259
+ turns,
260
+ loopback=loopback,
261
+ codec_profile=codec_profile,
262
+ seed=seed,
263
+ acoustic_operators=acoustic_operators,
264
+ )
265
+ # §2.5 binding correction: a deterministic in-process loopback is
266
+ # NEVER live_lane. Default codec round-trip is ON (9A-A11) → a stressed
267
+ # run → live_stressed; a no-op (codec_profile="none") clean run is also
268
+ # live_stressed at rung-2 (it never claims live_lane). captured_fixture
269
+ # is reached through the capture flow, not here.
270
+ evidence_class = "live_stressed"
271
+ elif rung != 1:
272
+ # rung == 3: unchanged keyed path; still requires the credentialed flag
273
+ # + RUNG3_REQUIRED_ENV; rung-3 lands as the owner live-proof (unit 7).
274
+ raise NotImplementedError(
275
+ f"livekit lane rung {rung} ({_RUNG_LABELS[rung]}) is not "
276
+ "implemented yet; rung 1 (virtual_clock) and rung 2 "
277
+ "(loopback_transport) are the supported tiers — rung 3 (cloud_sip) "
278
+ "is the owner-keyed live-proof lane"
279
+ )
280
+ else:
281
+ evidence_class = "live_stressed" if operators else "live_lane"
282
+
283
+ base_dir = (
284
+ Path(artifacts_dir)
285
+ if artifacts_dir is not None
286
+ else Path(tempfile.mkdtemp(prefix="agent-learning-live-livekit-"))
287
+ )
288
+ run_id = uuid.uuid4().hex
289
+ resolved_budget = float(budget_s) if budget_s is not None else lane_budget_s("livekit")
290
+ boot = {
291
+ "type": "boot",
292
+ "lane": "livekit",
293
+ "rung": rung,
294
+ "scenario": {"name": str(scenario.get("name") or "livekit-smoke")},
295
+ "turns": turns,
296
+ "config": {
297
+ "instructions": scenario.get("instructions")
298
+ or "You are a concise, helpful voice agent under test.",
299
+ "responses": scenario.get("responses"),
300
+ "expect": scenario.get("expect"),
301
+ },
302
+ }
303
+ worker = _WORKERS / "livekit_worker.py"
304
+
305
+ def _run_once(index: int, transcript: Any) -> dict[str, Any]:
306
+ return run_worker_once(
307
+ worker,
308
+ boot,
309
+ lane="livekit",
310
+ required_env=required,
311
+ cwd=base_dir,
312
+ timeout_s=resolved_budget,
313
+ transcript=transcript,
314
+ version_requirement=version_requirement,
315
+ )
316
+
317
+ result = run_repeated(
318
+ _run_once,
319
+ lane="livekit",
320
+ evidence_class=evidence_class,
321
+ repeats=repeats,
322
+ budget_s=budget_s,
323
+ required_env=required,
324
+ artifacts_dir=base_dir,
325
+ run_id=run_id,
326
+ rung=_RUNG_LABELS[rung],
327
+ framework="livekit-agents",
328
+ version_requirement=version_requirement,
329
+ )
330
+
331
+ events = primary_transcript_events(result)
332
+ # Normalization rides the existing realtime manifest builder — the run
333
+ # lands in the existing `realtime_trace` state family; the live engine
334
+ # is declared in metadata (guide §3.1).
335
+ from .. import simulate as _simulate
336
+
337
+ manifest = _simulate.build_realtime_run_manifest(
338
+ name=f"live-livekit-{run_id[:8]}",
339
+ framework="livekit",
340
+ required_env=required,
341
+ min_turns=1,
342
+ max_turns=max(len(turns), 1),
343
+ metadata={
344
+ "simulation_engine": "live_lane_livekit",
345
+ "live_lane": {"lane": "livekit", "rung": _RUNG_LABELS[rung]},
346
+ },
347
+ )
348
+
349
+ payload = lane_run_payload(
350
+ result,
351
+ name=f"live-livekit-{run_id[:8]}",
352
+ scenario=scenario,
353
+ manifest=manifest,
354
+ states={"realtime_trace": _realtime_state(events, rung_label=_RUNG_LABELS[rung])},
355
+ metadata={
356
+ "execution_model": "subprocess",
357
+ "rung": _RUNG_LABELS[rung],
358
+ # rung-1 honesty: timing-only voice metrics, NO channels block
359
+ "voice_timing": _voice_timing(events),
360
+ },
361
+ )
362
+ # the perturbations stanza carries BOTH families (text-rung records + the
363
+ # rung-2 acoustic records); the clean-twin link is filled by the campaign.
364
+ all_applied = list(applied) + list(acoustic_applied)
365
+ if all_applied:
366
+ payload["live_lane"]["perturbations"] = perturbations_stanza(
367
+ all_applied, seed=seed, paired_clean_run=None
368
+ )
369
+ if channels is not None:
370
+ # rung-2: attach the dual-channel evidence + the fidelity marker (§2.5 /
371
+ # 9A-A10). fidelity_tier is a MARKER FIELD, not a new evidence class.
372
+ payload["channels"] = channels
373
+ if isinstance(payload.get("live_lane"), dict):
374
+ payload["live_lane"]["fidelity_tier"] = fidelity_tier
375
+ payload["fidelity_tier"] = fidelity_tier
376
+ return payload
@@ -0,0 +1,172 @@
1
+ """MCP live lane (3E) — real MCP server processes over the real protocol.
2
+
3
+ Framework imports: NONE at module top (P3-D1). The default tier spawns the
4
+ shipped loopback stdio server (``_workers/mcp_loopback_server.py`` — a real
5
+ ``FastMCP`` process, credential-free but genuinely separate and speaking the
6
+ real protocol over the wire: that IS the live graduation, P3-D6/R§1 #12).
7
+ The client side is ``_workers/mcp_worker.py`` (a ``ClientSession`` over
8
+ stdio). Every artifact carries the server-behavior snapshot stamp
9
+ ``{server_name, server_version, capability_hash}`` (R§1 #11).
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import sys
15
+ import tempfile
16
+ import uuid
17
+ from pathlib import Path
18
+ from typing import Any, Mapping, Optional, Sequence
19
+
20
+ from ._contract import lane_budget_s, require_lane_enabled
21
+ from ._runner import run_worker_once
22
+ from ._stats import lane_run_payload, primary_transcript_events, run_repeated
23
+
24
+ _WORKERS = Path(__file__).resolve().parent / "_workers"
25
+ _RUNG_LABELS = {1: "loopback_servers", 2: "third_party_servers"}
26
+
27
+ # Deterministic, credential-free default tool script against the loopback
28
+ # server (claim-level expectations, tolerant of alternative trajectories).
29
+ _DEFAULT_CALLS = (
30
+ {"tool": "echo", "arguments": {"text": "hello loopback"}, "expect": {"contains": "hello loopback"}},
31
+ {"tool": "add", "arguments": {"a": 2, "b": 3}, "expect": {"contains": "5"}},
32
+ )
33
+
34
+
35
+ def _scenario_calls(scenario: Mapping[str, Any]) -> list[dict[str, Any]]:
36
+ raw = scenario.get("calls")
37
+ if not raw:
38
+ return [dict(call) for call in _DEFAULT_CALLS]
39
+ return [dict(call) for call in raw if isinstance(call, Mapping)]
40
+
41
+
42
+ def _server_snapshot(
43
+ events: Sequence[Mapping[str, Any]],
44
+ ) -> dict[str, Any] | None:
45
+ for event in events:
46
+ if event.get("type") == "server_snapshot":
47
+ payload = event.get("payload")
48
+ if isinstance(payload, Mapping):
49
+ return dict(payload)
50
+ return None
51
+
52
+
53
+ def _tool_session_state(events: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
54
+ items = []
55
+ for index, event in enumerate(events, start=1):
56
+ if event.get("channel") == "tool":
57
+ payload = event.get("payload")
58
+ payload = payload if isinstance(payload, Mapping) else {}
59
+ items.append(
60
+ {
61
+ "index": index,
62
+ "item_type": event.get("type"),
63
+ "tool": payload.get("name"),
64
+ "ok": payload.get("ok"),
65
+ }
66
+ )
67
+ return {
68
+ "engine": "live_lane_mcp",
69
+ "item_count": len(items),
70
+ "items": items[:200],
71
+ }
72
+
73
+
74
+ def run_mcp_lane(
75
+ scenario: Mapping[str, Any],
76
+ *,
77
+ server: Optional[Mapping[str, Any]] = None,
78
+ repeats: int = 8,
79
+ required_env: Optional[Sequence[str]] = None,
80
+ version_requirement: str | None = None,
81
+ budget_s: float | None = None,
82
+ artifacts_dir: str | Path | None = None,
83
+ ) -> dict[str, Any]:
84
+ """Default tier (server=None): loopback stdio server fixture + client.
85
+ Third-party tier (server={"command": [...], "env_names": [...]}) is
86
+ ``live_credentialed`` with server-specific names (P3-D6)."""
87
+
88
+ require_lane_enabled("mcp")
89
+ rung = 1 if server is None else 2
90
+ if rung >= 2:
91
+ require_lane_enabled("credentialed")
92
+
93
+ if server is None:
94
+ server_command = [sys.executable, str(_WORKERS / "mcp_loopback_server.py")]
95
+ server_env_names: list[str] = []
96
+ else:
97
+ command = server.get("command")
98
+ if not isinstance(command, Sequence) or not command:
99
+ raise ValueError(
100
+ "third-party server spec needs a non-empty 'command' list"
101
+ )
102
+ server_command = [str(part) for part in command]
103
+ server_env_names = [str(name) for name in server.get("env_names") or []]
104
+ required = tuple(
105
+ required_env if required_env is not None else server_env_names
106
+ )
107
+
108
+ base_dir = (
109
+ Path(artifacts_dir)
110
+ if artifacts_dir is not None
111
+ else Path(tempfile.mkdtemp(prefix="agent-learning-live-mcp-"))
112
+ )
113
+ run_id = uuid.uuid4().hex
114
+ resolved_budget = float(budget_s) if budget_s is not None else lane_budget_s("mcp")
115
+ boot = {
116
+ "type": "boot",
117
+ "lane": "mcp",
118
+ "rung": rung,
119
+ "scenario": {"name": str(scenario.get("name") or "mcp-loopback-smoke")},
120
+ "config": {
121
+ "server_command": server_command,
122
+ "server_env_names": server_env_names,
123
+ "calls": _scenario_calls(scenario),
124
+ },
125
+ }
126
+ worker = _WORKERS / "mcp_worker.py"
127
+
128
+ def _run_once(index: int, transcript: Any) -> dict[str, Any]:
129
+ return run_worker_once(
130
+ worker,
131
+ boot,
132
+ lane="mcp",
133
+ required_env=required,
134
+ cwd=base_dir,
135
+ timeout_s=resolved_budget,
136
+ transcript=transcript,
137
+ version_requirement=version_requirement,
138
+ )
139
+
140
+ result = run_repeated(
141
+ _run_once,
142
+ lane="mcp",
143
+ evidence_class="live_lane",
144
+ repeats=repeats,
145
+ budget_s=budget_s,
146
+ required_env=required,
147
+ artifacts_dir=base_dir,
148
+ run_id=run_id,
149
+ rung=_RUNG_LABELS[rung],
150
+ framework="mcp",
151
+ version_requirement=version_requirement,
152
+ )
153
+
154
+ events = primary_transcript_events(result)
155
+ payload = lane_run_payload(
156
+ result,
157
+ name=f"live-mcp-{run_id[:8]}",
158
+ scenario=scenario,
159
+ states={
160
+ "framework_runtime": {
161
+ "framework": "mcp",
162
+ "engine": "live_lane_mcp",
163
+ "rung": _RUNG_LABELS[rung],
164
+ },
165
+ "mcp_tool_session": _tool_session_state(events),
166
+ },
167
+ metadata={"execution_model": "subprocess", "rung": _RUNG_LABELS[rung]},
168
+ )
169
+ snapshot = _server_snapshot(events)
170
+ if snapshot is not None:
171
+ payload["live_lane"]["server_snapshot"] = snapshot
172
+ return payload