agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,127 @@
1
+ # Environment packaging conformance
2
+
3
+ The harness supports repository packaging incrementally while preserving one invariant: it may
4
+ package submitted code and provide infrastructure, but it never rewrites or invents agent tools.
5
+
6
+ ## Case 1: repository supplies Compose
7
+
8
+ ALK adopts the submitted Compose definition and:
9
+
10
+ - renders it with Compose rather than reimplementing its schema;
11
+ - rejects privileged/host escape features;
12
+ - assigns an isolated project, host ports and volumes;
13
+ - starts only default infrastructure services;
14
+ - identifies opt-in agent/worker services;
15
+ - discovers typed and generic TCP capabilities;
16
+ - injects external/internal endpoints through existing configuration seams;
17
+ - waits for Compose healthchecks, protocol-level readiness and a short startup-stability window;
18
+ - performs lifecycle reset with only that project's volumes; and
19
+ - tears down only that project.
20
+
21
+ Conformance fixture: `tests/fixtures/harness_agents/voice_analytics_agent`.
22
+
23
+ The real test starts two simultaneous copies even though the submitted Compose publishes fixed
24
+ ClickHouse and Redis ports. It starts both unchanged worker images, proves each reaches its own
25
+ seeded ClickHouse and Redis services, resets one project, verifies the other remains healthy, and
26
+ cleans up both.
27
+
28
+ ## Case 2: repository supplies Dockerfile but no Compose
29
+
30
+ ALK uses the submitted Dockerfile unchanged and generates a Compose adapter around it. A
31
+ standalone runtime is valid and gets no invented infrastructure. When the contract declares
32
+ dependencies, the adapter contains only those supported infrastructure services. Current managed
33
+ templates are:
34
+
35
+ | Engine | Connector | Reset mechanism |
36
+ |---|---|---|
37
+ | Postgres | `DATABASE_URL`/declared seam | project volume recreation + submitted SQL init |
38
+ | ClickHouse | `CLICKHOUSE_URL`/declared seam | project volume recreation + submitted SQL init |
39
+ | Redis | `REDIS_URL`/declared seam | project volume recreation |
40
+ | MongoDB | `MONGODB_URL`/declared seam | project volume recreation |
41
+ | Qdrant | `QDRANT_URL`/declared seam | project volume recreation |
42
+
43
+ Multiple dependencies are included in one private network and injected into the submitted
44
+ runtime. Generated services use no persisted resolved password, so the environment bundle passes
45
+ secret scanning and content sealing.
46
+
47
+ Conformance fixtures:
48
+
49
+ - `voice_analytics_agent` with Compose removed: managed ClickHouse + Redis;
50
+ - `voice_ledger_agent`: managed Postgres.
51
+
52
+ An additional generated Dockerfile-only conformance agent uses MongoDB and Qdrant together. Its
53
+ unchanged runtime writes both services, ALK destroys and recreates the project volumes, and a
54
+ second runtime proves both stores begin empty. The real integration gate completes in about 41
55
+ seconds on the current Docker Desktop host.
56
+
57
+ The public `dograh-hq/dograh` root Compose is also exercised unchanged. Its default API, UI,
58
+ pgvector/Postgres, password-protected Redis and MinIO services pass readiness and cleanup while
59
+ profile-gated TURN/init/tunnel services remain dormant. See `ENVIRONMENT_VALIDATION_MATRIX.md`
60
+ for evidence levels and the wider repository batch.
61
+
62
+ Both real tests build and run the submitted Dockerfile, query submitted seed data from inside the
63
+ runtime, emit application readiness evidence and clean up.
64
+
65
+ ### Official LiveKit agent validation
66
+
67
+ The opt-in real-agent suite also validates three unmodified agents from the official
68
+ `livekit/agents` examples repository at commit
69
+ `da6af86ac640a3bc54585764e64321d7048c1c16`:
70
+
71
+ - `drive_thru`: multi-tool ordering agent with in-process business state;
72
+ - `frontdesk`: scheduling agent with an optional external Cal.com integration and source-owned
73
+ fallback; and
74
+ - `hotel_receptionist`: larger multi-tool booking agent with SQLite-backed local state.
75
+
76
+ For each repository ALK detects a Dockerfile-only standalone runtime, builds the upstream image,
77
+ seals and verifies a portable environment bundle, starts the actual LiveKit worker with referenced
78
+ credentials, checks runtime health, verifies the source fingerprint did not change, and tears the
79
+ job down. No database or service is synthesized for these agents.
80
+
81
+ Credential discovery includes requirements consumed internally by a detected connector SDK. For
82
+ LiveKit that means `LIVEKIT_URL`, `LIVEKIT_API_KEY` and `LIVEKIT_API_SECRET` are requested before
83
+ worker startup even when submitted code contains no direct environment read. Resolved values are
84
+ passed through the child process environment and selected by name; they are never placed in
85
+ Docker command arguments or persisted in a bundle.
86
+
87
+ ## Explicit failure behavior
88
+
89
+ - A missing custom service/tool implementation is not generated.
90
+ - An unsupported engine is not replaced with a different database.
91
+ - External infrastructure with no Compose and no supported managed adapter fails before calls.
92
+ - A required runtime with no Compose and no Dockerfile fails with an actionable packaging error.
93
+ - Infrastructure/readiness failures are not agent evaluation results.
94
+
95
+ ## Running the matrix
96
+
97
+ Unit and contract checks:
98
+
99
+ ```bash
100
+ .venv/bin/pytest -q tests/test_harness_service_environments.py
101
+ ```
102
+
103
+ Real Docker checks:
104
+
105
+ ```bash
106
+ RUN_INTEGRATION=1 .venv/bin/pytest -q tests/test_harness_service_environments.py
107
+ ```
108
+
109
+ Official LiveKit example build and bundle checks:
110
+
111
+ ```bash
112
+ RUN_INTEGRATION=1 \
113
+ LIVEKIT_EXAMPLES_ROOT=/path/to/livekit-agents/examples \
114
+ .venv/bin/pytest -q tests/test_harness_livekit_examples.py -k builds
115
+ ```
116
+
117
+ Add `LIVEKIT_EXAMPLES_START_WORKERS=1` and provide the three discovered `LIVEKIT_*` values to
118
+ include real worker registration/readiness checks.
119
+
120
+ The real suite pulls/builds container images, starts services, mutates state, resets and removes
121
+ all test projects. It should run on CI with a dedicated Docker daemon, not a shared production
122
+ host.
123
+
124
+ Known protocols are checked semantically where possible (`/ping` for ClickHouse and `PING` for
125
+ Redis). This prevents a container or briefly-open socket from being reported ready while database
126
+ initialization is still restarting the service. Later liveness probes are point-in-time checks and
127
+ do not reapply the startup window.
@@ -0,0 +1,297 @@
1
+ # How the harness actually works
2
+
3
+ What happens between you typing a sentence and a graded result appearing. Written to be read
4
+ alongside the code, so every claim below names the file it lives in.
5
+
6
+ The shape is the same at every stage, and worth holding onto:
7
+
8
+ > **A stage is a model session with a small set of tools and its instructions in a markdown file.
9
+ > The model decides what to do; the tools do anything that must be exact and refuse anything that
10
+ > must not happen. Nothing reaches disk except through a tool that checked it first.**
11
+
12
+ There is no pipeline. Each stage is a conversation you can interrupt, correct, and resume.
13
+
14
+ ---
15
+
16
+ ## The pieces
17
+
18
+ | Piece | Where | What it is |
19
+ |---|---|---|
20
+ | Stage | `session.py` | A live model session, held open across turns, emitting typed events |
21
+ | Instructions | `skills/<stage>/SKILL.md` | How that stage works, in prose. Editable without touching code |
22
+ | Tools | `tools.py`, `world/tools.py`, `scenario_tools.py`, `run/tools.py` | The exact half: they execute, validate, and refuse |
23
+ | Artifacts | `artifacts/sessions/<id>/` | What each stage leaves behind for the next |
24
+ | Conversation | `chat.py` | Holds one agent's journey through the stages |
25
+ | Session | `sessions.py` | One conversation, one folder: chat and artifacts together |
26
+
27
+ The artifacts, each the input to the next stage:
28
+
29
+ ```
30
+ contract.json → world.sqlite + handlers/ + simulator_prompt.md + sub_goals.json
31
+ → scenarios/<name>/ (+ scenarios.json as the index) → runs.json
32
+ ```
33
+
34
+ All of it, plus the conversation that produced it, lives in one folder per session. There is
35
+ nothing held in memory that is not also on disk, so closing the page, restarting the server or
36
+ coming back tomorrow all resume the same way: by reading the folder.
37
+
38
+ ---
39
+
40
+ ## 1. Reception — which agent is this?
41
+
42
+ `reception.py`
43
+
44
+ A stage with `Read`, `Glob`, `Grep` and one tool, `point_at_agent`. You say where your agent
45
+ lives; it looks, confirms the path exists, picks a short name, and calls that tool.
46
+
47
+ It looks from the **workspace root** (the directory holding your repos), not from inside
48
+ `agent-learning-kit`, because the agent under test is almost never inside the harness.
49
+
50
+ `point_at(name, path, kind)` refuses a path that does not exist, and refuses an unknown kind.
51
+ `kind` selects an `AgentSource` from `sources.py` — `repo` (code on disk, gets file tools) or
52
+ `spec` (a prompt and tool schema pasted in, gets no file tools). **A new kind of agent is one
53
+ registration, not a new code path.**
54
+
55
+ ---
56
+
57
+ ## 2. Understand — what is this agent, verifiably?
58
+
59
+ `understand.py`, `skills/understand-agent/SKILL.md`, gate in `tools.py`
60
+
61
+ The session gets read-only file tools and one submission tool. It reads the agent's source and
62
+ calls `submit_contract`.
63
+
64
+ **The contract** (`contract.py`) is the anti-hallucination device for everything downstream:
65
+
66
+ | Field | Why it matters |
67
+ |---|---|
68
+ | `tools[]` — name, `args`, `arg_types`, `arg_values`, description | The agent's action space. `arg_values` are the real permitted values — the menu, the enum, the lookup |
69
+ | `hard_constraints[]` | Rules the agent must follow. Told to the agent under test, and graded by the judge |
70
+ | `base_environment` | Its real starting data, reproduced row for row |
71
+ | `real_use_cases[]` | What it is actually for |
72
+ | `notes` | Free-form: whatever else the reader judged worth carrying forward |
73
+ | `amendments[]` | Anything **not** read from source — see below |
74
+
75
+ **How it is written:** `accept_contract` in `tools.py` validates before anything reaches disk. It
76
+ refuses a contract with no tools, no use cases, duplicate tool names, types for arguments that do
77
+ not exist, or — the one that mattered most in practice — *every* tool having no arguments, which
78
+ means the arguments were read and then not recorded. Problems are returned **into the
79
+ conversation**, so the model corrects and resubmits rather than a bad contract landing.
80
+
81
+ **How it is changed later** (`amend.py`). The contract is not frozen, but every change is
82
+ recorded with a reason in `amendments[]`, so months later you can still tell what came from the
83
+ agent and what came from us:
84
+
85
+ - `amend_contract` — let an argument accept a value it did not before
86
+ - `add_rule` / `drop_rule` — a hard constraint the source did not state, or one misread
87
+ - `fix_tool` — correct argument names, types, description, or remove a tool that does not exist
88
+
89
+ Each demands a `why`. A contract that can be rewritten invisibly is no longer evidence.
90
+
91
+ ---
92
+
93
+ ## 3. Build — the world its tools run against
94
+
95
+ `build.py`, `skills/build-environment/SKILL.md`, tools in `world/tools.py`
96
+
97
+ **This is the part that makes the whole thing worth doing.** The repository's own services,
98
+ migrations, seed process and tool code run in an isolated environment. A call for something that
99
+ is not there is refused by the submitted implementation, not by a mock or a rewritten handler.
100
+
101
+ It builds three things, all shared by every scenario: **the world**, **the simulator prompt** for
102
+ a conversational agent, and **the sub-goal catalogue**. The stage has sixteen tools and no file
103
+ access at all:
104
+
105
+ | Tool | Does |
106
+ |---|---|
107
+ | `create_schema` | Run the CREATE TABLE statements |
108
+ | `seed` | Insert rows — the agent's real catalogue |
109
+ | `change_data` | One UPDATE or DELETE, for fixing a row put in wrong |
110
+ | `adopt_tool` | Bind and smoke-test the agent's own implementation |
111
+ | `write_env_file` / `run_env_command` | Container orchestration only; never agent behavior |
112
+ | `run_tool` | Call a defined tool and see what the world does |
113
+ | `declare_sequence` / `drop_sequence` | A series of calls whose end state must hold |
114
+ | `inspect_world` | Look at what is in the world |
115
+ | `amend_contract`, `add_rule`, `drop_rule`, `fix_tool` | Correct the contract |
116
+ | `check_world` | Run every probe, report without saving |
117
+ | `save_world` | Freeze it — refused unless it holds up |
118
+
119
+ Bindings are small adapters to the callable the repository ships. There is no generated-handler
120
+ fallback. If the callable cannot be imported or the shipped service cannot start, the build stops
121
+ and reports the missing runtime/configuration seam.
122
+
123
+ **The distinction the whole design turns on** (`runtime.py`):
124
+
125
+ - `ToolError` — the world saying *no*. The id does not exist; the item is unavailable. **This is
126
+ the world working.**
127
+ - Any other exception — our bug.
128
+
129
+ They are recorded differently and never confused. `_is_refusal` matches the agent's own refusal
130
+ type across the class hierarchy.
131
+
132
+ ### The gate: what `check_world` and `save_world` actually run
133
+
134
+ `world/probe.py`. Every probe restores the world to a frozen baseline first, so probes cannot
135
+ inherit each other's rows.
136
+
137
+ | Probe | Asks |
138
+ |---|---|
139
+ | `happy` | A valid call built from the contract's permitted values. A refusal is acceptable; a crash never is |
140
+ | `edge` | A nonexistent id → must refuse, not succeed and not crash. A missing required argument → must refuse |
141
+ | `coverage` | Every contract tool has a handler; no handler for a tool the agent lacks; **and each handler actually reads the arguments the contract says it takes** |
142
+ | `data` | Every identifier the contract permits exists in the world — catches a whole category left unseeded, which otherwise looks exactly like correct strictness |
143
+ | `sequence` | Each declared sequence, run from the frozen world, leaves the state it claims |
144
+ | unknown tool | Calling a tool that does not exist must refuse |
145
+
146
+ `save_world` refuses on three counts: **score below 0.85**; **no declared sequence** (calls that
147
+ each work alone can still forget what the last one did); and **a dirty world** — rows left over
148
+ from building, which would otherwise appear in every scenario as somebody else's order already
149
+ in the cart.
150
+
151
+ Then `world/snapshot.py` writes `world.sqlite`, `handlers/*.py`, `world.py` and `manifest.json`.
152
+ **The snapshot is the base state every scenario restores from.**
153
+
154
+ ---
155
+
156
+ ## 4. Scenarios — the conversations worth having
157
+
158
+ `scenarios.py`, `skills/write-scenarios/SKILL.md`, tools in `scenario_tools.py`
159
+
160
+ A scenario is a **change on the base environment**, and it owns a folder (`folder.py`):
161
+
162
+ ```
163
+ scenarios/<name>/
164
+ scenario.json the instruction, the reference solution, which sub-goals it names
165
+ setup.py def setup(world) what this scenario changes first
166
+ ready.py def ready(world) is the world ready for it
167
+ checks/<goal>.py def check(world, calls) one per deterministic sub-goal
168
+ ```
169
+
170
+ `scenarios.json` is an index over those folders, regenerated from them, so anything wanting the
171
+ whole suite at a glance has it.
172
+
173
+ Code lives in files, never duplicated into JSON, and the file is the artifact: each check file is
174
+ written with a `__main__` block so it runs standalone against what a run left behind, and a test
175
+ proves the file and the harness give the same answer.
176
+
177
+ `setup` is **code** rather than a list of rows: what a scenario changes is not necessarily the
178
+ database alone. There is no persona and no opening line. Variability comes from **real
179
+ conditions** that live in `setup`, and the base world stays the shared starting point.
180
+
181
+ `sub_goals` are **names from the catalogue the environment step defined**, not restated wording.
182
+ That is what makes results roll up: the same sub-goal failing in seven of twelve scenarios is one
183
+ sentence.
184
+
185
+ `solution` is what a correct agent would do. It is never run against the agent — it exists so the
186
+ scenario can be proved.
187
+
188
+ The writer can **look** (`inspect_world`) and **rehearse** (`try_calls` — run calls against a
189
+ throwaway copy and see what state they leave), so a solution is written from what was observed.
190
+
191
+ ### The gate: three proofs, no model involved
192
+
193
+ `prove.py`, called by `submit_scenario` before a scenario is kept:
194
+
195
+ | | Run | Must |
196
+ |---|---|---|
197
+ | **Ready** | reset → `setup` → `ready` | **hold** |
198
+ | **Solvable** | reset → setup → **the solution** → the checks | **pass** |
199
+ | **Not vacuous** | reset → setup → **nothing** → the checks | **fail** |
200
+
201
+ If the first fails, the world does not hold what the scenario presumes, and running it would test
202
+ us rather than the agent: the agent would fail for a precondition we got wrong, and it would read
203
+ as a finding about the agent. If the second fails, either the scenario cannot be passed or the
204
+ check is wrong. If the third passes, the checks grade nothing while reporting a result, which is
205
+ the failure that makes a suite quietly green.
206
+
207
+ Vacuity is judged on *all* checks passing, because one check surviving an empty run ("no
208
+ unavailable item was ordered") is legitimate. A single check that survives is still reported, in
209
+ `Proof.weak`: sub-goals are shared, so a check that cannot fail without calls would roll up as a
210
+ pass for an agent that did nothing at all.
211
+
212
+ `save_scenarios` additionally refuses a suite where no sub-goal is shared by two scenarios,
213
+ because nothing would roll up.
214
+
215
+ ---
216
+
217
+ ## 5. Run — put someone in front of it
218
+
219
+ `run/`
220
+
221
+ The simulation is **not a loop written here**. ALK already owns placing a call, driving the
222
+ synthetic user, and producing a transcript; the harness supplies the world, the instruction, and
223
+ the grading. Against the real hosted agent that is `run/live.py` and `run/call.py`:
224
+
225
+ ```
226
+ world + setup ──► webhook ──► public url ──► the assistant's OWN tools repointed
227
+ │
228
+ ALK's voice case places the call ──┘
229
+ │
230
+ the world afterwards + the calls ──► the sub-goals' checks
231
+ ```
232
+
233
+ 1. **Restore** the frozen world and apply this scenario's `setup`. Its own copy, so nothing leaks
234
+ between scenarios.
235
+ 2. **Stand up the webhook** (`run/voice.py`) and bind that world to it. A hosted voice agent
236
+ executes its tools by calling a webhook, so answering that webhook from `handle_tool_call` is
237
+ the entire integration.
238
+ 3. **Expose it** (`cloudflared`, or `HARNESS_WEBHOOK_URL`), because a hosted agent cannot reach
239
+ loopback.
240
+ 4. **Repoint the assistant.** `pointed_at` copies the agent's **own** tools and changes only
241
+ `server.url`. Nothing about the agent is redefined — rebuilding its tools would mean testing an
242
+ agent we wrote.
243
+ 5. **Place the call** through ALK's own voice case, with the scenario's filled simulator prompt
244
+ driving the caller.
245
+ 6. **Grade** from the world afterwards plus the recorded calls, through the same checks the gates
246
+ used. A sub-goal marked `judged` is reported as judged, never silently counted.
247
+
248
+ `run/alk.py` is the same story for the text path: the world goes in as `environment=` to ALK's
249
+ `ChatEnvironment`, which owns the turn loop.
250
+
251
+ Running is also a **stage of the conversation**, not only a command (`run/stage.py`,
252
+ `skills/run-scenarios/SKILL.md`, tools in `run/tools.py`: `preflight`, `list_scenarios`,
253
+ `run_scenario`, `read_results`). The stage exists because reading a failure is judgement: it has
254
+ to sort every failure into one of four causes — the agent was wrong, the world wrongly refused,
255
+ the check is wrong, or the simulated caller never asked for the thing — and only the first is a
256
+ finding about the agent. Each run's record lands in `runs.json` with the instruction, the
257
+ per-sub-goal verdicts, every tool call, and the transcript.
258
+
259
+ This is what the environment was built for. The previous run's known issues — *"mocked tools
260
+ always succeed, including removing an item that was never added"*, *"mock responses do not vary by
261
+ argument"*, *"world state does not change unless a scenario sets `state_updates`"* — are all the
262
+ same defect, and a world that really holds rows and can really refuse answers all three.
263
+
264
+ ---
265
+
266
+ ## What is exact, and what is judgement
267
+
268
+ The split is deliberate and worth defending:
269
+
270
+ | Judgement (the model) | Exact (code) |
271
+ |---|---|
272
+ | Reading unfamiliar source | Whether the contract is structurally usable |
273
+ | Designing a schema | Whether a handler crashes or refuses |
274
+ | Choosing what is worth testing | Whether an expectation resolves against real tables |
275
+ | Whether a claim held | Whether the state matches |
276
+
277
+ **The model never decides whether something passed.** It decides what to try.
278
+
279
+ ---
280
+
281
+ ## Where to extend it
282
+
283
+ - A new **agent kind** → a class in `sources.py` and one registration
284
+ - A new **world kind** (browser, filesystem, queue) → a class in `world/kinds.py` implementing
285
+ `values_present` / `mutable_state` / `describe`, and one registration. Browser is registered
286
+ and stubbed
287
+ - A new **place the agent runs** (Vapi, LiveKit, a hosted endpoint) → a class in `run/targets.py`
288
+ with `open` / `say` / `close`, whose tool calls reach the same `world.handle_tool_call`
289
+ - A change to **how a stage works** → edit its `SKILL.md`. No code
290
+
291
+ ## What is not built
292
+
293
+ - Browser worlds: registered, not implemented
294
+ - Snapshots are local files, not S3
295
+ - Judged sub-goals are reported as judged, not actually sent to a judge yet
296
+ - Results do not post to the platform
297
+ - Nothing reports which of the contract's use cases have no scenario
@@ -0,0 +1,229 @@
1
+ # Harness implementation and validation status
2
+
3
+ Status date: 22 August 2026
4
+
5
+ This document records what is implemented and what has actually been exercised. It complements
6
+ `ARCHITECTURE.md` (the target design) and `ENVIRONMENT_CONFORMANCE.md` (the packaging contract).
7
+
8
+ ## Product boundary now implemented
9
+
10
+ ALK owns the execution plane: repository understanding, environment construction, test-data
11
+ setup, scenario execution, evidence collection, grading and artifact production. The platform is
12
+ the control and data plane: it creates jobs, supplies job-scoped secret references, receives
13
+ events/results and presents history. Local CLI execution and hosted execution use the same
14
+ `HarnessJob`, `EnvironmentBundle`, executor and result contracts.
15
+
16
+ The central invariant is enforced throughout provisioning: ALK provides the environment in which
17
+ the submitted agent and its existing tools run. It does not rewrite tools, invent proprietary
18
+ service behavior or silently replace an unsupported dependency.
19
+
20
+ ## Completed implementation
21
+
22
+ ### Portable jobs, bundles and runtimes
23
+
24
+ - Immutable local and hosted job contracts, with local paths rejected for hosted jobs.
25
+ - Content-addressed environment bundles with provenance, per-file hashes and sizes.
26
+ - Bundle verification before execution; symlinks, path traversal and changed content are rejected.
27
+ - A runtime-provider boundary so local Docker/Compose can later be replaced by a hosted sandbox
28
+ provider without changing the harness pipeline.
29
+ - Structured lifecycle events, cancellation and terminal result delivery for platform-driven jobs.
30
+ - Retry classification that separates transient infrastructure/transport failures from agent
31
+ failures and avoids rerunning deterministic agent outcomes.
32
+
33
+ ### Repository and environment provisioning
34
+
35
+ - **Compose repository:** adopt the submitted Compose model, isolate project names, ports, networks
36
+ and volumes, start dependency services, inject test endpoints and clean up only that run.
37
+ - **Dockerfile-only repository:** keep the submitted runtime unchanged and generate only the
38
+ declared infrastructure around it. Managed adapters currently cover Postgres, ClickHouse,
39
+ Redis, MongoDB and Qdrant, including initialization, protocol-aware readiness and reset
40
+ behavior.
41
+ - **Neither Compose nor Dockerfile:** fail admission with an actionable packaging error when an
42
+ external runtime is required. Automatic packaging for this third case remains open work.
43
+ - Source fingerprints prevent reuse of stale builds.
44
+ - Protocol-aware readiness avoids treating a merely open socket as a ready database.
45
+
46
+ ### Credentials, GitHub input and isolation
47
+
48
+ - Static credential discovery from source, Compose, Dockerfile and known SDK connectors.
49
+ - Preflight reports missing credential names before a sandbox is started.
50
+ - Jobs and bundles contain `SecretRef` references, not resolved values; secret values are rejected
51
+ from persisted payloads and redacted from emitted output.
52
+ - GitHub repository/revision references are represented in the hosted job contract. The platform
53
+ remains responsible for repository authorization and job-scoped credential resolution.
54
+ - Unsafe Compose features such as privileged execution and host escape configuration are rejected.
55
+ - Resource/lifecycle ownership is scoped to the run so cleanup cannot target another environment.
56
+
57
+ ### Evidence and artifact integrity
58
+
59
+ - Setup activity is kept separate from agent activity and cannot earn evaluation credit.
60
+ - A missing tool call cannot satisfy a tool-dependent check; dependent checks cannot pass when the
61
+ prerequisite action is absent.
62
+ - Results distinguish environment, connectivity, simulator, agent, grading and artifact failures.
63
+ - Artifact manifests include hashes and sizes, are verified on ingestion and are delivered with
64
+ an idempotency identity for safe platform retries.
65
+ - LiveKit recording selection now ignores ambient/background publications and selects the actual
66
+ conversational speech track.
67
+
68
+ ## Environment conformance tested
69
+
70
+ The checked-in fixtures are small conformance repositories, not product demo agents:
71
+
72
+ | Packaging case | Dependencies | What was proven |
73
+ |---|---|---|
74
+ | Compose | ClickHouse + Redis | Two copies can run simultaneously despite fixed submitted ports; each worker reaches its own seeded services; resetting one does not affect the other. |
75
+ | Dockerfile only | ClickHouse + Redis | ALK composes managed dependencies around the unchanged image, injects endpoints and verifies seeded data from the running agent. |
76
+ | Dockerfile only | Postgres | ALK creates the database, applies submitted SQL, verifies application readiness and performs isolated cleanup. |
77
+ | Dockerfile only | MongoDB + Qdrant | ALK creates both services, the runtime writes/reads both, reset recreates clean state and cleanup removes the project. |
78
+
79
+ The real Docker conformance matrix exercised build, startup, readiness, state mutation, reset and
80
+ teardown. The detailed contract and commands are in `ENVIRONMENT_CONFORMANCE.md`.
81
+
82
+ ## Real voice-agent validation
83
+
84
+ Three agents from the official `livekit/agents` examples repository were validated at commit
85
+ `da6af86ac640a3bc54585764e64321d7048c1c16`:
86
+
87
+ - `drive_thru` — multi-tool ordering with in-process state;
88
+ - `frontdesk` — scheduling with an optional external Cal.com connector and source fallback; and
89
+ - `hotel_receptionist` — a larger booking flow with local SQLite state.
90
+
91
+ For all three, ALK detected the Dockerfile-only runtime, built the upstream image, produced and
92
+ verified a sealed bundle, started the worker, checked readiness and source integrity, and cleaned
93
+ up the execution environment.
94
+
95
+ Five real WebRTC calls were then made to each agent (15 total):
96
+
97
+ | Agent | Calls | Completed | Environment failures | Preserved agent failures |
98
+ |---|---:|---:|---:|---:|
99
+ | Drive-through | 5 | 5 | 0 | 0 |
100
+ | Front desk | 5 | 5 | 0 | 0 |
101
+ | Hotel receptionist | 5 | 4 | 0 | 1 |
102
+ | **Total** | **15** | **14** | **0** | **1** |
103
+
104
+ The hotel failure is a useful deterministic finding rather than a harness failure: the agent
105
+ repeatedly treated a valid 16-digit card number as 15 digits, entered a validation loop and reached
106
+ the seven-minute scenario limit.
107
+
108
+ Artifact checks passed for every call:
109
+
110
+ - 15/15 reports and transcripts were created;
111
+ - 15/15 combined recordings and 15/15 stereo recordings were present and larger than 100 KB;
112
+ - all recordings contained measurable audio; and
113
+ - all campaign containers were stopped after execution.
114
+
115
+ The public example repositories, temporary model-adapted checkouts, recordings and campaign
116
+ artifacts are intentionally not committed to ALK.
117
+
118
+ ### Hosted-provider connector validation
119
+
120
+ Two non-native target connectors were exercised with real calls using the unchanged configured
121
+ provider agents:
122
+
123
+ | Connector | Result | Evidence |
124
+ |---|---|---|
125
+ | Vapi WebSocket | Completed; evaluation 0.8401 | 14 committed messages, 1,071 transcript characters, SDK and provider recordings; combined audio peak -3.6 dB. |
126
+ | Retell Web Call | Completed; evaluation 0.8422 | 11 committed messages, 1,432 transcript characters, SDK and provider recordings; combined audio peak -1.9 dB. |
127
+
128
+ Both ended through `simulator_end_call`, produced no typed call failure and cleaned up their
129
+ FutureAGI LiveKit rooms. Daily and direct OpenAI Realtime remain separate transport gates; model
130
+ substitution must not be represented as transport validation.
131
+
132
+ ## Additional public-repository packaging validation
133
+
134
+ The environment admission path was also exercised against twelve structurally different public
135
+ voice-agent repositories on 22 August 2026. These checks intentionally used untouched checkouts;
136
+ repository defects remain findings instead of being patched by ALK.
137
+
138
+ | Repository | Revision | Packaging result | Runtime result |
139
+ |---|---|---|---|
140
+ | Bolna | `0172347b601ea66dac0414cc1c6b14dc0d85422a` | Nested `local_setup/docker-compose.yml` is now discovered. Host-bound AWS credential mounts are reported explicitly. | Not started: its local Compose requires provider/telephony credentials and a local `.env`; ALK correctly stops before build rather than fabricating them. |
141
+ | Pipecat examples (`websocket`) | `696c3541001350262dd11e81ccbf00754a56850c` | Root Dockerfile detected; `env.example` and its placeholder Google key are now recognized. | Rejected before Docker because the published Dockerfile requires an absent `uv.lock`. The same failure previously took about 80 seconds in BuildKit; static admission now reports it in about 1.2 seconds. |
142
+ | TEN Framework (`ai_agents`) | `2e56d9659d8599350962374c0dc24725a03d73ce` | Development Compose and production Dockerfile are distinguished; ALK selects the production Dockerfile and preserves the repository's `linux/amd64` hint. | The upstream release reaches `task install` but Bun exits 132 under Docker Desktop emulation on the ARM test host. This is recorded as host/runtime compatibility, not an environment success or agent evaluation. |
143
+ | Pipecat Cloud Starter | `9167c21ea76a67a02f330bae0c009d0ef5a6ef95` | Its root Dockerfile and sample credential declarations are accepted without modification. | The submitted image builds, but its real entrypoint fails: the archived repository leaves `pipecat-ai` unpinned and current 1.7.0 removed `LLMMessagesFrame`. Build success alone must not be represented as runtime success. A test-only compatibility checkout produced the real-call result below. |
144
+ | Voice Noob | `755552f1b92b3c760799720f2c250e7f0a6bf0bd` | Its Compose-provided Postgres and Redis are selected as infrastructure. Profile-gated pgAdmin and Redis Commander are not mistaken for agent runtimes. | Infrastructure only: the unchanged services started and passed readiness in 6.348 seconds, reset cleanly in 6.238 seconds and were removed. The repository does not package its FastAPI agent runtime in Compose or a Dockerfile; its browser uses direct OpenAI Realtime WebRTC, which the current LiveKit-only call engine cannot drive. This is not an end-to-end agent success. |
145
+ | Vocode Core (`telephony_app`) | `e054c33a72787b6a4920f91eb8598ad0bafb4240` | The component's Dockerfile and Compose are both found; the missing Compose `.env` is reported per affected service. | The unchanged Dockerfile reached dependency installation, then failed because it installs current Poetry while invoking the removed `--no-dev` option. This is an upstream reproducibility failure and is not patched by ALK. |
146
+ | Open Telephony Stack | `a42e5b6c17c772af72825dd40240807c415b220b` | The independently runnable Asterisk, shim and voice-server components are reported as an ambiguous monorepo rather than guessed. | The Asterisk Compose is rejected for hosted execution because it requests host networking and host TLS/system mounts. Its voice-server Dockerfile also requires the repository root as build context, proving the need for explicit component plus build-context selection. |
147
+ | Dograh | `058c540c4d92c55f529d04fabceb17da4901a0cb` | The root Compose runtime is selected; credential discovery is scoped to that runtime instead of every optional provider module. | The unchanged default stack started API, UI, pgvector/Postgres, authenticated Redis and MinIO in 36.391 seconds and cleaned up. This found and fixed authenticated-Redis readiness, empty-body HTTP readiness and dormant-profile port allocation. |
148
+ | Voice Asterisk Agent | `6a5e0533b5293d1727847d123bf2ab8a1fc136de` | Asterisk, AudioSocket, API, agent, STT/LLM/TTS and monitoring components are discovered without mistaking `Dockerfile.dockerignore` for a runtime. | Admission remains blocked by missing external env injection, component selection and forbidden Docker-socket access. These are real hosted-environment requirements, not an agent grade. |
149
+ | Aeyetech local voice agent | `34df18ab3e023bbf99b103bc4b891556999b6787` | Root Compose and local LiveKit/Whisper/llama.cpp/Kokoro/agent components are found. | Not started: a machine-specific absolute model bind mount and large CPU/GPU model requirements need explicit resource and mount admission first. |
150
+ | Salesforce VoiceAgentRAG | `4d653890db18855d564d4ed9ca4d678047799cf4` | Qdrant/FAISS and local/cloud model requirements are visible, but the repository has neither Compose nor a Dockerfile. | Correct case-three packaging finding; ALK does not silently invent a runtime. |
151
+ | LiveKit Node starter | `2045cc4917e40382fcf6b39e46ce24316a97370a` | Dockerfile-only Node runtime detected. Multi-source `COPY` validation now checks every input. | Rejected before Docker because the published Dockerfile requires absent `pnpm-lock.yaml`; the earlier BuildKit failure is now a deterministic admission finding. |
152
+
153
+ The new preflight is deliberately conservative. It detects missing Docker build inputs, ambiguous
154
+ monorepo component roots, nested Compose files, development-oriented Compose configuration,
155
+ privileged/host namespace requests and host bind mounts before starting a container. Large Docker
156
+ errors are bounded before entering job state so one failed build cannot overwhelm the platform UI.
157
+
158
+ ### Pipecat test-only real-call validation
159
+
160
+ Because no Daily credential was available, a temporary external checkout replaced Daily with
161
+ Pipecat's officially supported LiveKit transport. The archived starter's prompt and LLM behavior
162
+ were preserved. Deepgram supplied transcription and OpenAI replaced a Cartesia credential that
163
+ returned HTTP 401. These compatibility changes were not made to ALK or committed to the public
164
+ repository.
165
+
166
+ One real harness-driven WebRTC call then completed with 10 committed messages, 1,060 transcript
167
+ characters and `simulator_end_call`. Combined and stereo recordings were produced at 1,295,634
168
+ and 2,591,224 bytes; measured peak audio was -2.9 dB, proving that the files were not silence. The
169
+ target received and transcribed the caller audio, generated responses through its LLM pipeline and
170
+ returned synthesized speech through LiveKit. Its isolated runtime was removed after the call.
171
+
172
+ ### Test-only compatibility note
173
+
174
+ The current LiveKit server returned HTTP 401 for the example agents' LiveKit Inference models. To
175
+ exercise real calls, temporary external checkouts used direct Deepgram and OpenAI model providers
176
+ for STT/LLM/TTS. This was a test setup change only: no agent tools, prompts, database behavior or
177
+ harness source was rewritten. The temporary checkouts are not part of this branch.
178
+
179
+ ## Automated release gate
180
+
181
+ The harness-focused gate completed with:
182
+
183
+ ```text
184
+ 508 passed, 11 skipped
185
+ ```
186
+
187
+ The skipped cases are opt-in Docker/real-agent checks and require their documented runtime flags
188
+ and credentials. The WebRTC engine regression suite is included in the 508 passing tests. Ruff and
189
+ `git diff --check` also pass for this change set.
190
+
191
+ ## Findings that remain open
192
+
193
+ These items should not be represented as complete:
194
+
195
+ 1. Implement the production hosted provider (for example Daytona or a microVM fleet) behind the
196
+ existing runtime interface, with quotas, egress policy and hard cleanup guarantees.
197
+ 2. Add an explicit packaging adapter for repositories with neither Compose nor a Dockerfile.
198
+ 3. Expand managed dependency conformance beyond Postgres, ClickHouse, Redis, MongoDB and Qdrant.
199
+ RabbitMQ/NATS, MinIO/S3 and browser/code-execution services are the next explicit gates;
200
+ unsupported services must continue to fail explicitly.
201
+ 4. Add first-class component/build-context selection for monorepositories. Detection and
202
+ actionable ambiguity are implemented, but the platform does not yet let the user select one
203
+ of several independently runnable agents within a repository. Credential discovery now follows
204
+ an automatically selected Compose runtime and its build context, but must also follow an
205
+ explicit user-selected component in ambiguous monorepositories.
206
+ Compose files that require a repository-local `.env` also need a job-scoped external-env
207
+ adapter; ALK currently reports the missing file rather than mutating the submitted checkout.
208
+ Internal environment secrets such as a local stack's JWT signing key should be generated as
209
+ job-scoped secret references rather than requested from the customer.
210
+ 5. Add generic target-tool event ingestion for uninstrumented third-party LiveKit agents. Audio and
211
+ transcripts work today, but source-owned tool calls cannot always be normalized automatically.
212
+ Add target media adapters for non-LiveKit agents, beginning with direct OpenAI Realtime WebRTC,
213
+ Daily and generic bidirectional audio WebSocket. Voice Noob cannot be called through the current
214
+ engine until one of these adapters exists; switching its model does not change its transport.
215
+ 6. Productize GitHub authorization (GitHub App), secret storage/resolution and credential UX on the
216
+ platform. The ALK contracts and preflight are present; the production integrations are not.
217
+ 7. Improve platform run UX with granular live stages, logs, actionable waiting states and artifact
218
+ reconciliation rather than a long generic status.
219
+ 8. Continue work on diverse scenario/persona generation and controllable simulator behavior.
220
+ 9. Add production object-storage upload/reconciliation, retention policy and load/chaos testing.
221
+ 10. Track LiveKit transport instability observed as intermittent signal-resume 502/EOF responses.
222
+
223
+ ## Current conclusion
224
+
225
+ The local execution architecture and the environment creation path for well-packaged and
226
+ Dockerfile-only voice agents are functioning end to end. Real agents could be built, started and
227
+ called through the generated environment, and complete transcripts and audio artifacts were
228
+ produced. The remaining work is primarily the production hosted sandbox/provider integration,
229
+ broader packaging/service coverage, third-party tool telemetry and platform product experience.