agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,189 @@
1
+ """Pipecat lane worker (3C rung 1) — untrusted subprocess entry (P3-D1).
2
+
3
+ Builds a REAL Pipecat ``Pipeline`` and injects ``TranscriptionFrame``s
4
+ (bypassing STT/TTS — Pipecat's own documented eval technique), collecting
5
+ output text frames + TTFB timing into the JSONL stdio stream.
6
+
7
+ Boot config: ``pipeline_factory`` is an optional dotted ``module:attr``
8
+ returning a LIST of frame processors (the user's pipeline core); when
9
+ absent, a deterministic scripted responder is used. The worker always
10
+ appends its own collector sink to observe output frames.
11
+
12
+ IPC: see livekit_worker.py — same one-boot-line / JSONL-events contract.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import asyncio
18
+ import hashlib
19
+ import json
20
+ import os
21
+ import sys
22
+ import time
23
+ import traceback
24
+ from typing import Any
25
+
26
+
27
+ def _emit(channel: str, type_: str, payload: dict[str, Any]) -> None:
28
+ print(
29
+ json.dumps(
30
+ {"channel": channel, "type": type_, "payload": payload},
31
+ ensure_ascii=False,
32
+ default=str,
33
+ ),
34
+ flush=True,
35
+ )
36
+
37
+
38
+ def _read_boot() -> dict[str, Any]:
39
+ line = sys.stdin.readline()
40
+ if not line.strip():
41
+ raise RuntimeError("missing boot message on stdin")
42
+ boot = json.loads(line)
43
+ if not isinstance(boot, dict) or boot.get("type") != "boot":
44
+ raise RuntimeError("first stdin line must be a boot message")
45
+ return boot
46
+
47
+
48
+ def _capability_hash(framework: str, version: str) -> str:
49
+ return hashlib.sha256(f"{framework}:{version}".encode("utf-8")).hexdigest()
50
+
51
+
52
+ async def _run(boot: dict[str, Any]) -> None:
53
+ import importlib.metadata
54
+
55
+ import pipecat
56
+ from pipecat.frames.frames import EndFrame, TextFrame, TranscriptionFrame
57
+ from pipecat.pipeline.pipeline import Pipeline
58
+ from pipecat.pipeline.runner import PipelineRunner
59
+ from pipecat.pipeline.task import PipelineTask
60
+ from pipecat.processors.frame_processor import FrameDirection, FrameProcessor
61
+
62
+ version = importlib.metadata.version("pipecat-ai")
63
+ _emit(
64
+ "lane",
65
+ "framework_ready",
66
+ {
67
+ "framework": "pipecat-ai",
68
+ "framework_version": version,
69
+ "capability_hash": _capability_hash("pipecat-ai", version),
70
+ "package_paths": [os.path.dirname(pipecat.__file__)],
71
+ },
72
+ )
73
+ rung = int(boot.get("rung") or 1)
74
+ if rung != 1:
75
+ raise RuntimeError(f"pipecat worker implements rung 1 only, got {rung}")
76
+ config = boot.get("config") or {}
77
+ responses = [str(r) for r in (config.get("responses") or [])]
78
+ turns = boot.get("turns") or []
79
+
80
+ response_index = 0
81
+
82
+ class _ScriptedResponder(FrameProcessor):
83
+ """Deterministic stand-in for the user's LLM stage (rung 1)."""
84
+
85
+ async def process_frame(self, frame: Any, direction: Any) -> None:
86
+ nonlocal response_index
87
+ await super().process_frame(frame, direction)
88
+ if isinstance(frame, TranscriptionFrame):
89
+ if responses:
90
+ reply = responses[response_index % len(responses)]
91
+ else:
92
+ reply = f"ack: {frame.text}"
93
+ response_index += 1
94
+ await self.push_frame(TextFrame(reply), FrameDirection.DOWNSTREAM)
95
+ await self.push_frame(frame, direction)
96
+
97
+ collected: list[tuple[float, str]] = []
98
+
99
+ class _Collector(FrameProcessor):
100
+ async def process_frame(self, frame: Any, direction: Any) -> None:
101
+ await super().process_frame(frame, direction)
102
+ if isinstance(frame, TextFrame) and not isinstance(
103
+ frame, TranscriptionFrame
104
+ ):
105
+ collected.append((time.monotonic(), str(frame.text)))
106
+ await self.push_frame(frame, direction)
107
+
108
+ factory_path = config.get("pipeline_factory")
109
+ if factory_path:
110
+ module_name, _, attr = str(factory_path).partition(":")
111
+ if not module_name or not attr:
112
+ raise RuntimeError(
113
+ f"pipeline_factory must be 'module:attr', got {factory_path!r}"
114
+ )
115
+ factory = getattr(importlib.import_module(module_name), attr)
116
+ processors = factory()
117
+ if not isinstance(processors, (list, tuple)) or not processors:
118
+ raise RuntimeError(
119
+ "pipeline_factory must return a non-empty list of frame "
120
+ "processors (the worker appends its own collector sink)"
121
+ )
122
+ processors = list(processors)
123
+ else:
124
+ processors = [_ScriptedResponder()]
125
+ pipeline = Pipeline([*processors, _Collector()])
126
+ task = PipelineTask(pipeline)
127
+ runner = PipelineRunner(handle_sigint=False)
128
+
129
+ checks: list[bool] = []
130
+
131
+ async def _drive() -> None:
132
+ for index, turn in enumerate(turns):
133
+ text = str((turn or {}).get("user") or "")
134
+ _emit("user", "message", {"turn": index, "text": text})
135
+ injected_at = time.monotonic()
136
+ seen_before = len(collected)
137
+ frame_kwargs = {
138
+ "text": text,
139
+ "user_id": "user",
140
+ "timestamp": str(injected_at),
141
+ }
142
+ try:
143
+ frame = TranscriptionFrame(**frame_kwargs)
144
+ except TypeError:
145
+ frame = TranscriptionFrame(text, "user", str(injected_at))
146
+ await task.queue_frame(frame)
147
+ # Wait (bounded) for the pipeline to produce this turn's output.
148
+ deadline = time.monotonic() + 10.0
149
+ while len(collected) <= seen_before and time.monotonic() < deadline:
150
+ await asyncio.sleep(0.01)
151
+ new_outputs = collected[seen_before:]
152
+ if new_outputs:
153
+ first_at, reply = new_outputs[0]
154
+ ttfb_ms = round((first_at - injected_at) * 1000.0, 3)
155
+ _emit("agent", "message", {"turn": index, "text": reply})
156
+ _emit(
157
+ "lane",
158
+ "timing",
159
+ {"turn": index, "ttfb_ms": ttfb_ms},
160
+ )
161
+ ok = bool(reply.strip())
162
+ expect = (turn or {}).get("expect")
163
+ if isinstance(expect, dict) and isinstance(
164
+ expect.get("contains"), str
165
+ ):
166
+ ok = ok and expect["contains"].lower() in reply.lower()
167
+ checks.append(ok)
168
+ else:
169
+ checks.append(False)
170
+ await task.queue_frame(EndFrame())
171
+
172
+ await asyncio.gather(runner.run(task), _drive())
173
+ passed = bool(checks) and all(checks)
174
+ _emit("lane", "verification", {"passed": passed, "checks": checks})
175
+
176
+
177
+ def main() -> int:
178
+ boot = _read_boot()
179
+ try:
180
+ asyncio.run(_run(boot))
181
+ except Exception:
182
+ _emit("lane", "worker_error", {"traceback": traceback.format_exc()})
183
+ traceback.print_exc(file=sys.stderr)
184
+ return 1
185
+ return 0
186
+
187
+
188
+ if __name__ == "__main__":
189
+ sys.exit(main())
@@ -0,0 +1,138 @@
1
+ """A2A live lane (3E) — one adapter over heterogeneous A2A peers.
2
+
3
+ Framework imports: NONE at module top (P3-D1). The default tier is a
4
+ loopback peer pair: ``_workers/a2a_worker.py`` in client mode spawns its own
5
+ peer-mode sibling on 127.0.0.1 and walks the protocol stages — card
6
+ discovery → task lifecycle → artifact exchange (R§1 #18). Remote peers are
7
+ ``live_credentialed``. Live red-team scenarios point the existing corpus at
8
+ these targets; there is NO separate red-team marker.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import tempfile
14
+ import uuid
15
+ from pathlib import Path
16
+ from typing import Any, Mapping, Optional, Sequence
17
+
18
+ from ._contract import lane_budget_s, require_lane_enabled
19
+ from ._runner import run_worker_once
20
+ from ._stats import lane_run_payload, primary_transcript_events, run_repeated
21
+
22
+ _WORKERS = Path(__file__).resolve().parent / "_workers"
23
+ _RUNG_LABELS = {1: "loopback_peers", 2: "external_peers"}
24
+
25
+ _DEFAULT_STAGES = ("card_discovery", "task_lifecycle", "artifact_exchange")
26
+
27
+
28
+ def _scenario_stages(scenario: Mapping[str, Any]) -> list[str]:
29
+ raw = scenario.get("stages")
30
+ if not raw:
31
+ return list(_DEFAULT_STAGES)
32
+ stages = [str(stage) for stage in raw if str(stage) in _DEFAULT_STAGES]
33
+ return stages or list(_DEFAULT_STAGES)
34
+
35
+
36
+ def _protocol_state(events: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
37
+ items = []
38
+ for index, event in enumerate(events, start=1):
39
+ if event.get("channel") in ("agent", "tool", "user"):
40
+ payload = event.get("payload")
41
+ payload = payload if isinstance(payload, Mapping) else {}
42
+ items.append(
43
+ {
44
+ "index": index,
45
+ "channel": event.get("channel"),
46
+ "item_type": event.get("type"),
47
+ "stage": payload.get("stage"),
48
+ "ok": payload.get("ok"),
49
+ }
50
+ )
51
+ return {
52
+ "engine": "live_lane_a2a",
53
+ "item_count": len(items),
54
+ "items": items[:200],
55
+ }
56
+
57
+
58
+ def run_a2a_lane(
59
+ scenario: Mapping[str, Any],
60
+ *,
61
+ peer: Optional[str] = None,
62
+ repeats: int = 8,
63
+ required_env: Optional[Sequence[str]] = None,
64
+ version_requirement: str | None = None,
65
+ budget_s: float | None = None,
66
+ artifacts_dir: str | Path | None = None,
67
+ ) -> dict[str, Any]:
68
+ """Default tier (peer=None): loopback peer pair. A remote peer URL is
69
+ the ``live_credentialed`` tier."""
70
+
71
+ require_lane_enabled("a2a")
72
+ rung = 1 if peer is None else 2
73
+ if rung >= 2:
74
+ require_lane_enabled("credentialed")
75
+
76
+ required = tuple(required_env) if required_env is not None else ()
77
+ base_dir = (
78
+ Path(artifacts_dir)
79
+ if artifacts_dir is not None
80
+ else Path(tempfile.mkdtemp(prefix="agent-learning-live-a2a-"))
81
+ )
82
+ run_id = uuid.uuid4().hex
83
+ resolved_budget = float(budget_s) if budget_s is not None else lane_budget_s("a2a")
84
+ boot = {
85
+ "type": "boot",
86
+ "lane": "a2a",
87
+ "rung": rung,
88
+ "mode": "client",
89
+ "scenario": {"name": str(scenario.get("name") or "a2a-loopback-smoke")},
90
+ "config": {
91
+ "peer_url": peer,
92
+ "stages": _scenario_stages(scenario),
93
+ "message": str(scenario.get("message") or "ping from the harness"),
94
+ },
95
+ }
96
+ worker = _WORKERS / "a2a_worker.py"
97
+
98
+ def _run_once(index: int, transcript: Any) -> dict[str, Any]:
99
+ return run_worker_once(
100
+ worker,
101
+ boot,
102
+ lane="a2a",
103
+ required_env=required,
104
+ cwd=base_dir,
105
+ timeout_s=resolved_budget,
106
+ transcript=transcript,
107
+ version_requirement=version_requirement,
108
+ )
109
+
110
+ result = run_repeated(
111
+ _run_once,
112
+ lane="a2a",
113
+ evidence_class="live_lane",
114
+ repeats=repeats,
115
+ budget_s=budget_s,
116
+ required_env=required,
117
+ artifacts_dir=base_dir,
118
+ run_id=run_id,
119
+ rung=_RUNG_LABELS[rung],
120
+ framework="a2a-sdk",
121
+ version_requirement=version_requirement,
122
+ )
123
+
124
+ events = primary_transcript_events(result)
125
+ return lane_run_payload(
126
+ result,
127
+ name=f"live-a2a-{run_id[:8]}",
128
+ scenario=scenario,
129
+ states={
130
+ "framework_runtime": {
131
+ "framework": "a2a",
132
+ "engine": "live_lane_a2a",
133
+ "rung": _RUNG_LABELS[rung],
134
+ },
135
+ "protocol_trace": _protocol_state(events),
136
+ },
137
+ metadata={"execution_model": "subprocess", "rung": _RUNG_LABELS[rung]},
138
+ )
@@ -0,0 +1,339 @@
1
+ """LangChain/LangGraph live lane (3D) — real compiled graphs, checkpoints.
2
+
3
+ Two execution paths, selected by what the caller passes (P3-D1):
4
+
5
+ - **In-process** when the caller passes a live Python graph object (a
6
+ ``CompiledStateGraph``) — the existing ``wrap_agent`` contract users
7
+ already accept. Framework access happens through the object the caller
8
+ built; any framework import here is lazy, inside function bodies only.
9
+ - **Subprocess** via ``_workers/langgraph_worker.py`` when the lane boots
10
+ from a factory path (a dotted ``module:factory`` string): the worker
11
+ imports the factory, compiles the graph, and runs the same turn script
12
+ under the scrubbed-env subprocess model. The artifact records which
13
+ execution model ran.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import tempfile
19
+ import traceback as _traceback
20
+ import uuid
21
+ from pathlib import Path
22
+ from typing import Any, Mapping, Optional, Sequence
23
+
24
+ from ._contract import lane_budget_s, require_lane_enabled
25
+ from ._runner import run_worker_once, version_preflight
26
+ from ._stats import (
27
+ lane_run_payload,
28
+ primary_transcript_events,
29
+ run_repeated,
30
+ step_signature_from_events,
31
+ )
32
+
33
+ _WORKERS = Path(__file__).resolve().parent / "_workers"
34
+ _RUNG_LABELS = {1: "scripted_local_model", 2: "credentialed_model"}
35
+
36
+ _DEFAULT_TURNS = (
37
+ {"user": "Hello - what can you do?"},
38
+ {"user": "Summarize our conversation so far."},
39
+ )
40
+
41
+
42
+ def _scenario_turns(scenario: Mapping[str, Any]) -> list[dict[str, Any]]:
43
+ raw = scenario.get("turns") or scenario.get("user_messages")
44
+ if not raw:
45
+ return [dict(turn) for turn in _DEFAULT_TURNS]
46
+ turns: list[dict[str, Any]] = []
47
+ for item in raw:
48
+ if isinstance(item, str):
49
+ turns.append({"user": item})
50
+ elif isinstance(item, Mapping):
51
+ turns.append(dict(item))
52
+ return turns or [dict(turn) for turn in _DEFAULT_TURNS]
53
+
54
+
55
+ def _langgraph_version() -> str | None:
56
+ try:
57
+ import importlib.metadata
58
+
59
+ return importlib.metadata.version("langgraph")
60
+ except Exception:
61
+ return None
62
+
63
+
64
+ def _turn_input(turn: Mapping[str, Any]) -> Any:
65
+ if "input" in turn:
66
+ return turn["input"]
67
+ return {"messages": [{"role": "user", "content": str(turn.get("user") or "")}]}
68
+
69
+
70
+ def _last_message_text(output: Any) -> str:
71
+ if isinstance(output, Mapping):
72
+ messages = output.get("messages")
73
+ if isinstance(messages, Sequence) and messages:
74
+ last = messages[-1]
75
+ content = getattr(last, "content", None)
76
+ if content is None and isinstance(last, Mapping):
77
+ content = last.get("content")
78
+ if content is not None:
79
+ return str(content)
80
+ return str(output)
81
+ return str(output)
82
+
83
+
84
+ def _turn_check(turn: Mapping[str, Any], reply: str) -> bool:
85
+ expect = turn.get("expect")
86
+ if isinstance(expect, Mapping) and isinstance(expect.get("contains"), str):
87
+ return expect["contains"].lower() in reply.lower()
88
+ return bool(reply.strip())
89
+
90
+
91
+ def _workflow_state(
92
+ events: Sequence[Mapping[str, Any]], *, execution_model: str
93
+ ) -> dict[str, Any]:
94
+ items = []
95
+ for index, event in enumerate(events, start=1):
96
+ if event.get("channel") in ("user", "agent", "tool"):
97
+ payload = event.get("payload")
98
+ payload = payload if isinstance(payload, Mapping) else {}
99
+ items.append(
100
+ {
101
+ "index": index,
102
+ "channel": event.get("channel"),
103
+ "item_type": event.get("type"),
104
+ "text": payload.get("text"),
105
+ }
106
+ )
107
+ return {
108
+ "engine": "live_lane_langgraph",
109
+ "execution_model": execution_model,
110
+ "item_count": len(items),
111
+ "items": items[:200],
112
+ }
113
+
114
+
115
+ def run_langgraph_lane(
116
+ graph_or_factory: Any, # CompiledStateGraph object → in-process;
117
+ # "pkg.module:make_graph" → subprocess
118
+ # via _workers/langgraph_worker.py (P3-D1)
119
+ scenario: Mapping[str, Any],
120
+ *,
121
+ repeats: int = 8,
122
+ checkpointer: Any | None = None, # in-process: a live checkpointer object;
123
+ # subprocess: "memory" | "sqlite"
124
+ cross_session_probe: bool = True,
125
+ rung: int = 1,
126
+ required_env: Optional[Sequence[str]] = None,
127
+ version_requirement: str | None = None,
128
+ budget_s: float | None = None,
129
+ artifacts_dir: str | Path | None = None,
130
+ ) -> dict[str, Any]:
131
+ require_lane_enabled("langchain")
132
+ if rung >= 2:
133
+ require_lane_enabled("credentialed")
134
+ if rung not in _RUNG_LABELS:
135
+ raise ValueError(f"rung must be one of {sorted(_RUNG_LABELS)}, got {rung}")
136
+
137
+ required = tuple(required_env) if required_env is not None else ()
138
+ turns = _scenario_turns(scenario)
139
+ base_dir = (
140
+ Path(artifacts_dir)
141
+ if artifacts_dir is not None
142
+ else Path(tempfile.mkdtemp(prefix="agent-learning-live-langgraph-"))
143
+ )
144
+ run_id = uuid.uuid4().hex
145
+ resolved_budget = (
146
+ float(budget_s) if budget_s is not None else lane_budget_s("langchain")
147
+ )
148
+ subprocess_path = isinstance(graph_or_factory, str)
149
+ execution_model = "subprocess" if subprocess_path else "in_process"
150
+
151
+ if subprocess_path:
152
+ if checkpointer is not None and not isinstance(checkpointer, str):
153
+ raise ValueError(
154
+ "the subprocess (factory) path takes checkpointer as a string "
155
+ "('memory' or 'sqlite'); live checkpointer objects cannot "
156
+ "cross the process boundary"
157
+ )
158
+ boot = {
159
+ "type": "boot",
160
+ "lane": "langchain",
161
+ "rung": rung,
162
+ "scenario": {"name": str(scenario.get("name") or "langgraph-smoke")},
163
+ "turns": turns,
164
+ "config": {
165
+ "factory": graph_or_factory,
166
+ "checkpointer": checkpointer or "memory",
167
+ "cross_session_probe": bool(cross_session_probe),
168
+ "probe": scenario.get("probe"),
169
+ "thread_id": f"live-{run_id[:8]}",
170
+ },
171
+ }
172
+ worker = _WORKERS / "langgraph_worker.py"
173
+
174
+ def _run_once(index: int, transcript: Any) -> dict[str, Any]:
175
+ return run_worker_once(
176
+ worker,
177
+ boot,
178
+ lane="langchain",
179
+ required_env=required,
180
+ cwd=base_dir,
181
+ timeout_s=resolved_budget,
182
+ transcript=transcript,
183
+ version_requirement=version_requirement,
184
+ )
185
+
186
+ else:
187
+ graph = graph_or_factory
188
+
189
+ def _run_once(index: int, transcript: Any) -> dict[str, Any]:
190
+ # In-process path: the caller's live graph object, the accepted
191
+ # wrap_agent contract. Verification is programmatic per turn.
192
+ version = _langgraph_version()
193
+ preflight = version_preflight(
194
+ version_requirement,
195
+ {
196
+ "framework": "langgraph",
197
+ "framework_version": version,
198
+ "capability_hash": None,
199
+ },
200
+ )
201
+ transcript.record(
202
+ "lane",
203
+ "framework_ready",
204
+ {
205
+ "framework": "langgraph",
206
+ "framework_version": version,
207
+ "capability_hash": None,
208
+ "package_paths": [],
209
+ "execution_model": "in_process",
210
+ },
211
+ )
212
+ row: dict[str, Any] = {
213
+ "transcript_path": str(transcript.path),
214
+ "version": preflight,
215
+ }
216
+ if not preflight["version_ok"]:
217
+ row.update(
218
+ passed=None,
219
+ score=None,
220
+ failure_layer="lane_infra",
221
+ void_reason=preflight["void_reason"],
222
+ detail=str(preflight["void_reason"]),
223
+ )
224
+ return row
225
+ thread_id = f"live-{run_id[:8]}-r{index}"
226
+ config = {"configurable": {"thread_id": thread_id}}
227
+ checks: list[bool] = []
228
+ try:
229
+ for turn_index, turn in enumerate(turns):
230
+ transcript.record(
231
+ "user",
232
+ "message",
233
+ {"turn": turn_index, "text": str(turn.get("user") or "")},
234
+ )
235
+ output = graph.invoke(_turn_input(turn), config=config)
236
+ reply = _last_message_text(output)
237
+ transcript.record(
238
+ "agent", "message", {"turn": turn_index, "text": reply}
239
+ )
240
+ checks.append(_turn_check(turn, reply))
241
+ probe = scenario.get("probe")
242
+ if cross_session_probe and isinstance(probe, Mapping):
243
+ # Same-object cross-session probe: state must survive a
244
+ # second session on the same thread. The full
245
+ # discard-and-rebuild probe needs a factory — that is
246
+ # the subprocess path's job (guide §3.3).
247
+ inject = str(probe.get("inject") or "")
248
+ question = str(probe.get("question") or "What do you remember?")
249
+ if inject:
250
+ transcript.record(
251
+ "user", "message", {"session": 1, "text": inject}
252
+ )
253
+ graph.invoke(_turn_input({"user": inject}), config=config)
254
+ transcript.record(
255
+ "user", "message", {"session": 2, "text": question}
256
+ )
257
+ output = graph.invoke(_turn_input({"user": question}), config=config)
258
+ reply = _last_message_text(output)
259
+ transcript.record(
260
+ "agent", "message", {"session": 2, "text": reply}
261
+ )
262
+ fired = (
263
+ str(probe.get("assert_contains") or "").lower()
264
+ in reply.lower()
265
+ if probe.get("assert_contains")
266
+ else bool(reply.strip())
267
+ )
268
+ contained = (
269
+ str(probe.get("assert_not_contains") or "").lower()
270
+ not in reply.lower()
271
+ if probe.get("assert_not_contains")
272
+ else True
273
+ )
274
+ checks.append(fired and contained)
275
+ transcript.record(
276
+ "lane",
277
+ "cross_session_probe",
278
+ {
279
+ "probe_mode": "same_object",
280
+ "fired": fired,
281
+ "contained": contained,
282
+ },
283
+ )
284
+ except Exception as exc:
285
+ transcript.record(
286
+ "lane",
287
+ "worker_error",
288
+ {"traceback": _traceback.format_exc()},
289
+ )
290
+ row.update(
291
+ passed=False,
292
+ score=0.0,
293
+ failure_layer="framework_runtime",
294
+ detail=f"graph invoke raised: {exc}",
295
+ step_signature=step_signature_from_events(transcript.events),
296
+ )
297
+ return row
298
+ passed = bool(checks) and all(checks)
299
+ transcript.record(
300
+ "lane", "verification", {"passed": passed, "checks": checks}
301
+ )
302
+ row.update(
303
+ passed=passed,
304
+ score=1.0 if passed else 0.0,
305
+ failure_layer=None if passed else "agent_behavior",
306
+ detail="" if passed else "programmatic turn checks failed",
307
+ step_signature=step_signature_from_events(transcript.events),
308
+ )
309
+ return row
310
+
311
+ result = run_repeated(
312
+ _run_once,
313
+ lane="langchain",
314
+ evidence_class="live_lane",
315
+ repeats=repeats,
316
+ budget_s=budget_s,
317
+ required_env=required,
318
+ artifacts_dir=base_dir,
319
+ run_id=run_id,
320
+ rung=_RUNG_LABELS[rung],
321
+ framework="langgraph",
322
+ version_requirement=version_requirement,
323
+ )
324
+
325
+ events = primary_transcript_events(result)
326
+ payload = lane_run_payload(
327
+ result,
328
+ name=f"live-langgraph-{run_id[:8]}",
329
+ scenario=scenario,
330
+ states={
331
+ "workflow_trace": _workflow_state(events, execution_model=execution_model)
332
+ },
333
+ metadata={
334
+ "execution_model": execution_model,
335
+ "rung": _RUNG_LABELS[rung],
336
+ "cross_session_probe": bool(cross_session_probe),
337
+ },
338
+ )
339
+ return payload