agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,217 @@
1
+ """LangGraph lane worker (3D, manifest/factory path) — untrusted subprocess
2
+ entry (P3-D1).
3
+
4
+ Imports the caller's ``module:factory``, compiles the REAL graph against a
5
+ real checkpoint store (MemorySaver or SqliteSaver in the run tempdir), runs
6
+ the turn script via ``invoke`` on the same thread_id, and executes the
7
+ cross-session probe (R§1 #6): session 1 injects via the persistence channel,
8
+ the graph object is DISCARDED and REBUILT against the same checkpointer,
9
+ session 2 asserts firing/containment on the same thread. End-state diffs of
10
+ the checkpoint store are emitted as ``lane/end_state_diff`` (R§1 #14).
11
+
12
+ IPC: see livekit_worker.py — same one-boot-line / JSONL-events contract.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ import json
19
+ import os
20
+ import sys
21
+ import traceback
22
+ from typing import Any
23
+
24
+
25
+ def _emit(channel: str, type_: str, payload: dict[str, Any]) -> None:
26
+ print(
27
+ json.dumps(
28
+ {"channel": channel, "type": type_, "payload": payload},
29
+ ensure_ascii=False,
30
+ default=str,
31
+ ),
32
+ flush=True,
33
+ )
34
+
35
+
36
+ def _read_boot() -> dict[str, Any]:
37
+ line = sys.stdin.readline()
38
+ if not line.strip():
39
+ raise RuntimeError("missing boot message on stdin")
40
+ boot = json.loads(line)
41
+ if not isinstance(boot, dict) or boot.get("type") != "boot":
42
+ raise RuntimeError("first stdin line must be a boot message")
43
+ return boot
44
+
45
+
46
+ def _capability_hash(framework: str, version: str) -> str:
47
+ return hashlib.sha256(f"{framework}:{version}".encode("utf-8")).hexdigest()
48
+
49
+
50
+ def _package_paths(module: Any) -> list[str]:
51
+ """Filesystem roots of a framework package for traceback attribution.
52
+
53
+ langgraph (like livekit) ships as a NAMESPACE package: ``__file__`` is
54
+ None and the roots live on ``__path__`` instead.
55
+ """
56
+
57
+ file = getattr(module, "__file__", None)
58
+ if file:
59
+ return [os.path.dirname(file)]
60
+ return [str(path) for path in getattr(module, "__path__", None) or []]
61
+
62
+
63
+ def _turn_input(turn: dict[str, Any]) -> Any:
64
+ if "input" in turn:
65
+ return turn["input"]
66
+ return {"messages": [{"role": "user", "content": str(turn.get("user") or "")}]}
67
+
68
+
69
+ def _last_message_text(output: Any) -> str:
70
+ if isinstance(output, dict):
71
+ messages = output.get("messages")
72
+ if isinstance(messages, (list, tuple)) and messages:
73
+ last = messages[-1]
74
+ content = getattr(last, "content", None)
75
+ if content is None and isinstance(last, dict):
76
+ content = last.get("content")
77
+ if content is not None:
78
+ return str(content)
79
+ return str(output)
80
+ return str(output)
81
+
82
+
83
+ def _run(boot: dict[str, Any]) -> None:
84
+ import importlib.metadata
85
+
86
+ import langgraph
87
+
88
+ version = importlib.metadata.version("langgraph")
89
+ package_paths = _package_paths(langgraph)
90
+ try:
91
+ import langchain_core
92
+
93
+ package_paths.extend(_package_paths(langchain_core))
94
+ except ImportError:
95
+ pass
96
+ _emit(
97
+ "lane",
98
+ "framework_ready",
99
+ {
100
+ "framework": "langgraph",
101
+ "framework_version": version,
102
+ "capability_hash": _capability_hash("langgraph", version),
103
+ "package_paths": package_paths,
104
+ "execution_model": "subprocess",
105
+ },
106
+ )
107
+ config = boot.get("config") or {}
108
+ factory_path = str(config.get("factory") or "")
109
+ module_name, _, attr = factory_path.partition(":")
110
+ if not module_name or not attr:
111
+ raise RuntimeError(f"factory must be 'module:attr', got {factory_path!r}")
112
+ factory = getattr(importlib.import_module(module_name), attr)
113
+
114
+ checkpointer_kind = str(config.get("checkpointer") or "memory")
115
+ if checkpointer_kind == "sqlite":
116
+ import sqlite3
117
+
118
+ from langgraph.checkpoint.sqlite import SqliteSaver
119
+
120
+ connection = sqlite3.connect("checkpoints.sqlite", check_same_thread=False)
121
+ checkpointer = SqliteSaver(connection)
122
+ else:
123
+ from langgraph.checkpoint.memory import MemorySaver
124
+
125
+ checkpointer = MemorySaver()
126
+
127
+ def _build_graph() -> Any:
128
+ try:
129
+ candidate = factory(checkpointer=checkpointer)
130
+ except TypeError:
131
+ candidate = factory()
132
+ if hasattr(candidate, "compile"):
133
+ candidate = candidate.compile(checkpointer=checkpointer)
134
+ return candidate
135
+
136
+ thread_id = str(config.get("thread_id") or "live-thread")
137
+ invoke_config = {"configurable": {"thread_id": thread_id}}
138
+
139
+ def _checkpoint_count() -> int | None:
140
+ try:
141
+ return sum(1 for _ in checkpointer.list(invoke_config))
142
+ except Exception:
143
+ return None
144
+
145
+ graph = _build_graph()
146
+ checkpoints_before = _checkpoint_count()
147
+ checks: list[bool] = []
148
+ turns = boot.get("turns") or []
149
+ for index, turn in enumerate(turns):
150
+ turn = turn or {}
151
+ text = str(turn.get("user") or "")
152
+ _emit("user", "message", {"turn": index, "text": text, "session": 1})
153
+ output = graph.invoke(_turn_input(turn), config=invoke_config)
154
+ reply = _last_message_text(output)
155
+ _emit("agent", "message", {"turn": index, "text": reply, "session": 1})
156
+ expect = turn.get("expect")
157
+ ok = bool(reply.strip())
158
+ if isinstance(expect, dict) and isinstance(expect.get("contains"), str):
159
+ ok = ok and expect["contains"].lower() in reply.lower()
160
+ checks.append(ok)
161
+
162
+ probe = config.get("probe")
163
+ if config.get("cross_session_probe") and isinstance(probe, dict):
164
+ inject = str(probe.get("inject") or "")
165
+ question = str(probe.get("question") or "What do you remember?")
166
+ if inject:
167
+ _emit("user", "message", {"session": 1, "text": inject, "probe": True})
168
+ graph.invoke(_turn_input({"user": inject}), config=invoke_config)
169
+ # Discard and REBUILD against the same checkpointer — the process
170
+ # crosses a real persistence boundary, not an in-memory alias.
171
+ del graph
172
+ graph = _build_graph()
173
+ _emit("user", "message", {"session": 2, "text": question, "probe": True})
174
+ output = graph.invoke(_turn_input({"user": question}), config=invoke_config)
175
+ reply = _last_message_text(output)
176
+ _emit("agent", "message", {"session": 2, "text": reply, "probe": True})
177
+ fired = True
178
+ if isinstance(probe.get("assert_contains"), str):
179
+ fired = probe["assert_contains"].lower() in reply.lower()
180
+ contained = True
181
+ if isinstance(probe.get("assert_not_contains"), str):
182
+ contained = probe["assert_not_contains"].lower() not in reply.lower()
183
+ _emit(
184
+ "lane",
185
+ "cross_session_probe",
186
+ {"probe_mode": "rebuilt", "fired": fired, "contained": contained},
187
+ )
188
+ checks.append(fired and contained)
189
+
190
+ checkpoints_after = _checkpoint_count()
191
+ _emit(
192
+ "lane",
193
+ "end_state_diff",
194
+ {
195
+ "checkpoint_store": checkpointer_kind,
196
+ "checkpoints_before": checkpoints_before,
197
+ "checkpoints_after": checkpoints_after,
198
+ "thread_id": thread_id,
199
+ },
200
+ )
201
+ passed = bool(checks) and all(checks)
202
+ _emit("lane", "verification", {"passed": passed, "checks": checks})
203
+
204
+
205
+ def main() -> int:
206
+ boot = _read_boot()
207
+ try:
208
+ _run(boot)
209
+ except Exception:
210
+ _emit("lane", "worker_error", {"traceback": traceback.format_exc()})
211
+ traceback.print_exc(file=sys.stderr)
212
+ return 1
213
+ return 0
214
+
215
+
216
+ if __name__ == "__main__":
217
+ sys.exit(main())
@@ -0,0 +1,207 @@
1
+ """LiveKit lane worker (3B rung 1) — untrusted subprocess entry (P3-D1).
2
+
3
+ Boots a REAL ``livekit.agents.AgentSession`` and drives it with the
4
+ first-party text-rung helper ``session.run(user_input=...)`` (LiveKit's own
5
+ pytest surface) under a virtual-clock turn script with a deterministic
6
+ scripted LLM (no transport, no credentials — P3-D3 rung 1).
7
+
8
+ IPC: reads ONE boot JSON line on stdin; emits one JSON object per line on
9
+ stdout: ``{"channel": "user"|"agent"|"tool"|"lane", "type": ..., "payload": ...}``.
10
+ The handshake event is ``lane/framework_ready`` carrying
11
+ ``{framework, framework_version, capability_hash, package_paths}``.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import asyncio
17
+ import hashlib
18
+ import json
19
+ import os
20
+ import sys
21
+ import traceback
22
+ from typing import Any
23
+
24
+
25
+ def _emit(channel: str, type_: str, payload: dict[str, Any]) -> None:
26
+ print(
27
+ json.dumps(
28
+ {"channel": channel, "type": type_, "payload": payload},
29
+ ensure_ascii=False,
30
+ default=str,
31
+ ),
32
+ flush=True,
33
+ )
34
+
35
+
36
+ def _read_boot() -> dict[str, Any]:
37
+ line = sys.stdin.readline()
38
+ if not line.strip():
39
+ raise RuntimeError("missing boot message on stdin")
40
+ boot = json.loads(line)
41
+ if not isinstance(boot, dict) or boot.get("type") != "boot":
42
+ raise RuntimeError("first stdin line must be a boot message")
43
+ return boot
44
+
45
+
46
+ def _capability_hash(framework: str, version: str) -> str:
47
+ return hashlib.sha256(f"{framework}:{version}".encode("utf-8")).hexdigest()
48
+
49
+
50
+ def _extract_reply(result: Any) -> str:
51
+ """Pull the last assistant text out of a session.run RunResult,
52
+ defensively across livekit-agents 1.x minor versions."""
53
+
54
+ texts: list[str] = []
55
+ for event in getattr(result, "events", None) or []:
56
+ item = getattr(event, "item", None)
57
+ if getattr(item, "role", None) != "assistant":
58
+ continue
59
+ text_content = getattr(item, "text_content", None)
60
+ if text_content:
61
+ texts.append(str(text_content))
62
+ continue
63
+ content = getattr(item, "content", None)
64
+ if isinstance(content, str):
65
+ texts.append(content)
66
+ elif isinstance(content, (list, tuple)):
67
+ texts.append(
68
+ " ".join(str(part) for part in content if isinstance(part, str))
69
+ )
70
+ return texts[-1] if texts else ""
71
+
72
+
73
+ async def _run(boot: dict[str, Any]) -> None:
74
+ import importlib.metadata
75
+
76
+ import livekit
77
+ from livekit.agents import Agent, AgentSession
78
+ from livekit.agents import llm as lk_llm
79
+
80
+ version = importlib.metadata.version("livekit-agents")
81
+ _emit(
82
+ "lane",
83
+ "framework_ready",
84
+ {
85
+ "framework": "livekit-agents",
86
+ "framework_version": version,
87
+ "capability_hash": _capability_hash("livekit-agents", version),
88
+ # livekit is a NAMESPACE package: __file__ is None, roots are on
89
+ # __path__ (same fix as langgraph_worker._package_paths).
90
+ "package_paths": (
91
+ [os.path.dirname(livekit.__file__)]
92
+ if getattr(livekit, "__file__", None)
93
+ else [str(path) for path in getattr(livekit, "__path__", None) or []]
94
+ ),
95
+ },
96
+ )
97
+ rung = int(boot.get("rung") or 1)
98
+ if rung != 1:
99
+ raise RuntimeError(f"livekit worker implements rung 1 only, got {rung}")
100
+ config = boot.get("config") or {}
101
+ responses = [str(r) for r in (config.get("responses") or [])]
102
+ instructions = str(
103
+ config.get("instructions")
104
+ or "You are a concise, helpful voice agent under test."
105
+ )
106
+ expect = config.get("expect") if isinstance(config.get("expect"), dict) else {}
107
+ turns = boot.get("turns") or []
108
+
109
+ def _make_chunk(text: str) -> Any:
110
+ try:
111
+ return lk_llm.ChatChunk(
112
+ id="scripted",
113
+ delta=lk_llm.ChoiceDelta(role="assistant", content=text),
114
+ )
115
+ except TypeError:
116
+ return lk_llm.ChatChunk(
117
+ request_id="scripted",
118
+ choices=[
119
+ lk_llm.Choice(
120
+ delta=lk_llm.ChoiceDelta(role="assistant", content=text),
121
+ index=0,
122
+ )
123
+ ],
124
+ )
125
+
126
+ class _ScriptedStream(lk_llm.LLMStream):
127
+ def __init__(self, llm_obj: Any, *, chat_ctx: Any, tools: Any, conn_options: Any, text: str) -> None:
128
+ super().__init__(
129
+ llm_obj, chat_ctx=chat_ctx, tools=tools, conn_options=conn_options
130
+ )
131
+ self._text = text
132
+
133
+ async def _run(self) -> None:
134
+ self._event_ch.send_nowait(_make_chunk(self._text))
135
+
136
+ default_conn_options = getattr(lk_llm, "DEFAULT_API_CONNECT_OPTIONS", None)
137
+ if default_conn_options is None:
138
+ try:
139
+ from livekit.agents.types import DEFAULT_API_CONNECT_OPTIONS as default_conn_options
140
+ except ImportError:
141
+ default_conn_options = None
142
+
143
+ class _ScriptedLLM(lk_llm.LLM):
144
+ """Deterministic stub LLM node — rung 1 is credential-free (P3-D3)."""
145
+
146
+ def __init__(self) -> None:
147
+ super().__init__()
148
+ self._index = 0
149
+
150
+ @property
151
+ def model(self) -> str:
152
+ return "scripted-stub"
153
+
154
+ def chat(self, *, chat_ctx: Any, tools: Any = None, conn_options: Any = None, **kwargs: Any) -> Any:
155
+ if responses:
156
+ text = responses[self._index % len(responses)]
157
+ else:
158
+ text = "Acknowledged."
159
+ self._index += 1
160
+ return _ScriptedStream(
161
+ self,
162
+ chat_ctx=chat_ctx,
163
+ tools=tools or [],
164
+ conn_options=conn_options or default_conn_options,
165
+ text=text,
166
+ )
167
+
168
+ session = AgentSession(llm=_ScriptedLLM())
169
+ await session.start(Agent(instructions=instructions))
170
+ checks: list[bool] = []
171
+ try:
172
+ for index, turn in enumerate(turns):
173
+ text = str((turn or {}).get("user") or "")
174
+ _emit("user", "message", {"turn": index, "text": text})
175
+ result = await session.run(user_input=text)
176
+ reply = _extract_reply(result)
177
+ _emit("agent", "message", {"turn": index, "text": reply})
178
+ ok = bool(reply.strip())
179
+ contains = (turn or {}).get("expect", {}).get("contains") if isinstance((turn or {}).get("expect"), dict) else None
180
+ contains = contains or expect.get("contains")
181
+ if isinstance(contains, str):
182
+ ok = ok and contains.lower() in reply.lower()
183
+ checks.append(ok)
184
+ finally:
185
+ aclose = getattr(session, "aclose", None)
186
+ if aclose is not None:
187
+ try:
188
+ await aclose()
189
+ except Exception:
190
+ pass
191
+ passed = bool(checks) and all(checks)
192
+ _emit("lane", "verification", {"passed": passed, "checks": checks})
193
+
194
+
195
+ def main() -> int:
196
+ boot = _read_boot()
197
+ try:
198
+ asyncio.run(_run(boot))
199
+ except Exception:
200
+ _emit("lane", "worker_error", {"traceback": traceback.format_exc()})
201
+ traceback.print_exc(file=sys.stderr)
202
+ return 1
203
+ return 0
204
+
205
+
206
+ if __name__ == "__main__":
207
+ sys.exit(main())
@@ -0,0 +1,46 @@
1
+ """Loopback MCP stdio server (P3-D6) — the shipped credential-free fixture.
2
+
3
+ A REAL ``FastMCP`` server process with deterministic tools: credential-free
4
+ but a genuinely separate process speaking the real protocol over the wire —
5
+ that IS the live graduation (R§1 #12). Spawned by ``mcp_worker.py`` via the
6
+ MCP SDK's own stdio transport; its stdio IS the MCP wire (it does not speak
7
+ the lane JSONL protocol).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import sys
13
+
14
+ SERVER_NAME = "agent-learning-loopback"
15
+ SERVER_VERSION = "1.0.0"
16
+
17
+
18
+ def main() -> int:
19
+ from mcp.server.fastmcp import FastMCP
20
+
21
+ server = FastMCP(SERVER_NAME)
22
+
23
+ @server.tool()
24
+ def echo(text: str) -> str:
25
+ """Echo the input text back verbatim."""
26
+
27
+ return text
28
+
29
+ @server.tool()
30
+ def add(a: float, b: float) -> float:
31
+ """Add two numbers deterministically."""
32
+
33
+ return a + b
34
+
35
+ @server.tool()
36
+ def sort_unique(items: list[str]) -> list[str]:
37
+ """Return the sorted, de-duplicated items (deterministic)."""
38
+
39
+ return sorted(set(items))
40
+
41
+ server.run("stdio")
42
+ return 0
43
+
44
+
45
+ if __name__ == "__main__":
46
+ sys.exit(main())
@@ -0,0 +1,158 @@
1
+ """MCP lane worker (3E) — untrusted subprocess entry (P3-D1).
2
+
3
+ The client side of the MCP lane: a REAL ``ClientSession`` over the SDK's
4
+ stdio transport. It launches the target server (default: the shipped
5
+ loopback fixture ``mcp_loopback_server.py``) as ITS OWN subprocess via
6
+ ``StdioServerParameters`` — the real protocol over the wire between two
7
+ separate processes — lists tools (capability hash = sha256 of the sorted
8
+ ``list_tools()`` JSON, R§1 #11), runs the scenario's tool-call script with
9
+ claim-level expectations, and emits the server-behavior snapshot stamp.
10
+
11
+ IPC with the harness: see livekit_worker.py — same one-boot-line / JSONL
12
+ contract on THIS process's stdio (the MCP wire is the server subprocess's
13
+ stdio, owned by the SDK).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import asyncio
19
+ import hashlib
20
+ import json
21
+ import os
22
+ import sys
23
+ import traceback
24
+ from typing import Any
25
+
26
+
27
+ def _emit(channel: str, type_: str, payload: dict[str, Any]) -> None:
28
+ print(
29
+ json.dumps(
30
+ {"channel": channel, "type": type_, "payload": payload},
31
+ ensure_ascii=False,
32
+ default=str,
33
+ ),
34
+ flush=True,
35
+ )
36
+
37
+
38
+ def _read_boot() -> dict[str, Any]:
39
+ line = sys.stdin.readline()
40
+ if not line.strip():
41
+ raise RuntimeError("missing boot message on stdin")
42
+ boot = json.loads(line)
43
+ if not isinstance(boot, dict) or boot.get("type") != "boot":
44
+ raise RuntimeError("first stdin line must be a boot message")
45
+ return boot
46
+
47
+
48
+ def _result_text(result: Any) -> str:
49
+ texts: list[str] = []
50
+ for block in getattr(result, "content", None) or []:
51
+ text = getattr(block, "text", None)
52
+ if text:
53
+ texts.append(str(text))
54
+ return "\n".join(texts)
55
+
56
+
57
+ async def _run(boot: dict[str, Any]) -> None:
58
+ import importlib.metadata
59
+
60
+ import mcp as mcp_pkg
61
+ from mcp import ClientSession, StdioServerParameters
62
+ from mcp.client.stdio import stdio_client
63
+
64
+ version = importlib.metadata.version("mcp")
65
+ config = boot.get("config") or {}
66
+ command = [str(part) for part in (config.get("server_command") or [])]
67
+ if not command:
68
+ command = [
69
+ sys.executable,
70
+ os.path.join(os.path.dirname(os.path.abspath(__file__)), "mcp_loopback_server.py"),
71
+ ]
72
+ env_names = [str(name) for name in (config.get("server_env_names") or [])]
73
+ server_env = {name: os.environ[name] for name in env_names if name in os.environ}
74
+ # The server inherits the kit path so the loopback fixture imports clean.
75
+ if "PYTHONPATH" in os.environ:
76
+ server_env.setdefault("PYTHONPATH", os.environ["PYTHONPATH"])
77
+ if "PATH" in os.environ:
78
+ server_env.setdefault("PATH", os.environ["PATH"])
79
+
80
+ params = StdioServerParameters(
81
+ command=command[0], args=command[1:], env=server_env or None
82
+ )
83
+ async with stdio_client(params) as (read_stream, write_stream):
84
+ async with ClientSession(read_stream, write_stream) as session:
85
+ init = await session.initialize()
86
+ tools_result = await session.list_tools()
87
+ tool_summary = sorted(
88
+ (
89
+ {
90
+ "name": tool.name,
91
+ "description": tool.description or "",
92
+ }
93
+ for tool in tools_result.tools
94
+ ),
95
+ key=lambda item: item["name"],
96
+ )
97
+ capability_hash = hashlib.sha256(
98
+ json.dumps(tool_summary, sort_keys=True).encode("utf-8")
99
+ ).hexdigest()
100
+ _emit(
101
+ "lane",
102
+ "framework_ready",
103
+ {
104
+ "framework": "mcp",
105
+ "framework_version": version,
106
+ "capability_hash": capability_hash,
107
+ "package_paths": [os.path.dirname(mcp_pkg.__file__)],
108
+ },
109
+ )
110
+ server_info = getattr(init, "serverInfo", None)
111
+ _emit(
112
+ "lane",
113
+ "server_snapshot",
114
+ {
115
+ "server_name": getattr(server_info, "name", None),
116
+ "server_version": getattr(server_info, "version", None),
117
+ "capability_hash": capability_hash,
118
+ },
119
+ )
120
+ checks: list[bool] = []
121
+ for call in config.get("calls") or []:
122
+ call = call or {}
123
+ name = str(call.get("tool") or "")
124
+ arguments = call.get("arguments") or {}
125
+ _emit("tool", "tool_call", {"name": name, "arguments": arguments})
126
+ result = await session.call_tool(name, arguments)
127
+ text = _result_text(result)
128
+ ok = not bool(getattr(result, "isError", False))
129
+ # Claim-level rubric, tolerant of alternative trajectories
130
+ # (R§1 #11): the claim is about the answer, not the path.
131
+ expect = call.get("expect")
132
+ if isinstance(expect, dict) and isinstance(
133
+ expect.get("contains"), str
134
+ ):
135
+ ok = ok and expect["contains"].lower() in text.lower()
136
+ _emit(
137
+ "tool",
138
+ "tool_result",
139
+ {"name": name, "ok": ok, "text": text[:2000]},
140
+ )
141
+ checks.append(ok)
142
+ passed = bool(checks) and all(checks)
143
+ _emit("lane", "verification", {"passed": passed, "checks": checks})
144
+
145
+
146
+ def main() -> int:
147
+ boot = _read_boot()
148
+ try:
149
+ asyncio.run(_run(boot))
150
+ except Exception:
151
+ _emit("lane", "worker_error", {"traceback": traceback.format_exc()})
152
+ traceback.print_exc(file=sys.stderr)
153
+ return 1
154
+ return 0
155
+
156
+
157
+ if __name__ == "__main__":
158
+ sys.exit(main())