agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/extensions.py ADDED
@@ -0,0 +1,163 @@
1
+ """Unit 4 (BBG U4 / ARCH §2e) — the four 13D-4 registries + extension_admission.
2
+
3
+ Explicit in-process registration (AD-J: no entry-points, no import-time
4
+ discovery — local-first, gate-scannable). One uniform record shape across four
5
+ points; ONE choke point (``extension_admission``) enforces 13D-D5 (extension ≠
6
+ exemption) in gated contexts. The facade pushes world-kind/role registrations
7
+ DOWN into ``fi.simulate.simulation.contract`` setters (Appendix C-1); the engine
8
+ never imports up.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from typing import Any, Dict, Mapping, Optional
13
+
14
+ from .live._contract import EVIDENCE_CLASSES, RELEASE_ADMISSIBLE_EVIDENCE_CLASSES
15
+
16
+ EXTENSION_POINTS = ("environment", "loss", "optimizer", "generator")
17
+
18
+ # point -> name -> record
19
+ _REGISTRY: Dict[str, Dict[str, dict]] = {point: {} for point in EXTENSION_POINTS}
20
+
21
+
22
+ class ExtensionError(ValueError):
23
+ """Raised when a registration record is malformed."""
24
+
25
+
26
+ def _validate_record(point: str, record: Mapping[str, Any]) -> dict:
27
+ if point not in EXTENSION_POINTS:
28
+ raise ExtensionError(f"point {point!r} not in {EXTENSION_POINTS}")
29
+ name = record.get("name")
30
+ if not name or "." not in str(name):
31
+ raise ExtensionError(
32
+ "extension record requires a namespaced name 'vendor.name'"
33
+ )
34
+ if str(name) in _REGISTRY[point]:
35
+ raise ExtensionError(f"extension name collision: {name!r} already registered for {point}")
36
+ caps = record.get("evidence_class_capability") or []
37
+ for cap in caps:
38
+ if cap not in EVIDENCE_CLASSES:
39
+ raise ExtensionError(
40
+ f"evidence_class_capability {cap!r} not in {EVIDENCE_CLASSES}"
41
+ )
42
+ if point == "optimizer" and not record.get("declared_budgets"):
43
+ raise ExtensionError(
44
+ "optimizer extensions REQUIRE declared_budgets (refused otherwise)"
45
+ )
46
+ stored = dict(record)
47
+ stored["point"] = point
48
+ stored.setdefault("provides", None)
49
+ stored.setdefault("conformance_manifest", None)
50
+ stored.setdefault("evidence_class_capability", list(caps))
51
+ stored.setdefault("version", "0.0.0")
52
+ stored["gated_contexts_runnable"] = False # until a green conformance run lands
53
+ return stored
54
+
55
+
56
+ def register_extension(point: str, record: Mapping[str, Any]) -> dict:
57
+ stored = _validate_record(point, record)
58
+ # world-kind/role registrations push down into the engine contract (C-1).
59
+ if point == "environment":
60
+ token = stored.get("kind_token")
61
+ if token:
62
+ if not stored.get("spec_validator") or not stored.get("rung_ladder"):
63
+ raise ExtensionError(
64
+ "a custom world.kind extension MUST declare spec_validator + rung_ladder (R4)"
65
+ )
66
+ from fi.simulate.simulation import contract as _contract
67
+ _contract.register_world_kind(str(token), stored)
68
+ _REGISTRY[point][str(stored["name"])] = stored
69
+ return stored
70
+
71
+
72
+ def register_environment(record: Mapping[str, Any]) -> dict:
73
+ """Register a descriptive environment **metadata record** (studio extension).
74
+
75
+ Canon correspondence (assessment §8 Gap B): the runtime sibling
76
+ ``fi.simulate.registry.register_environment`` registers a **runnable** plugin
77
+ factory in ``environment_registry``. This one records metadata (and, for a
78
+ ``world.kind`` extension carrying a ``kind_token``, writes the contract's
79
+ extension side-table via ``contract.register_world_kind`` — the frozen canon
80
+ constants never mutate). A record is not a factory — the two are deliberately
81
+ unwired.
82
+ """
83
+ return register_extension("environment", record)
84
+
85
+
86
+ def register_objective(record: Mapping[str, Any]) -> dict:
87
+ return register_extension("loss", record)
88
+
89
+
90
+ def register_optimizer(record: Mapping[str, Any]) -> dict:
91
+ return register_extension("optimizer", record)
92
+
93
+
94
+ def register_generator(record: Mapping[str, Any]) -> dict:
95
+ return register_extension("generator", record)
96
+
97
+
98
+ def register_role(record: Mapping[str, Any]) -> dict:
99
+ """Register a namespaced cast role (pushes into the engine contract)."""
100
+ stored = _validate_record("environment", {**record, "name": record["name"]})
101
+ token = stored.get("kind_token") or stored.get("role")
102
+ from fi.simulate.simulation import contract as _contract
103
+ if token:
104
+ _contract.register_cast_role(str(token), stored)
105
+ _REGISTRY["environment"][str(stored["name"])] = stored
106
+ return stored
107
+
108
+
109
+ def resolve(point: str, token: str) -> Optional[dict]:
110
+ """Built-ins first — callers check the canon BEFORE consulting this."""
111
+ return _REGISTRY.get(point, {}).get(str(token))
112
+
113
+
114
+ def registered(point: str) -> tuple[str, ...]:
115
+ return tuple(sorted(_REGISTRY.get(point, {})))
116
+
117
+
118
+ def extension_admission(record: Mapping[str, Any], context: Mapping[str, Any]) -> dict:
119
+ """THE one choke point. In gated contexts (release-check/promotion/training)
120
+ a registered extension that cannot produce admissible evidence does not run.
121
+ Returns ``{admitted: bool}`` or the structured refusal finding."""
122
+ gated = bool(context.get("gated"))
123
+ if not gated:
124
+ return {"admitted": True, "reason": "non_gated_passthrough"}
125
+
126
+ point = record.get("point")
127
+ caps = set(record.get("evidence_class_capability") or [])
128
+ admissible_caps = caps & set(RELEASE_ADMISSIBLE_EVIDENCE_CLASSES)
129
+ conformance_green = bool(record.get("conformance_green") or record.get("gated_contexts_runnable"))
130
+
131
+ def refuse(reason: str) -> dict:
132
+ return {
133
+ "admitted": False,
134
+ "type": "extension_evidence_inadmissible",
135
+ "level": "error",
136
+ "name": record.get("name"),
137
+ "point": point,
138
+ "reason": reason,
139
+ "remediation": (
140
+ "run the extension's conformance_manifest green with evidence in "
141
+ f"{RELEASE_ADMISSIBLE_EVIDENCE_CLASSES} before any gated use"
142
+ ),
143
+ }
144
+
145
+ # (i) conformance manifest ran green with release-admissible evidence
146
+ if not conformance_green or not admissible_caps:
147
+ return refuse("no green conformance run with release-admissible evidence")
148
+ # (iii) optimizer extensions without declared_budgets never run
149
+ if point == "optimizer" and not record.get("declared_budgets"):
150
+ return refuse("optimizer extension has no declared_budgets")
151
+ # (iv) environment/world-kind extensions claiming an executable kind need a
152
+ # green rung-1 fixture run recorded.
153
+ if point == "environment" and record.get("kind_token"):
154
+ if not record.get("rung1_fixture_green"):
155
+ return refuse("world-kind extension lacks a green rung-1 fixture run")
156
+ return {"admitted": True, "reason": "gated_admitted"}
157
+
158
+
159
+ def _reset_extensions() -> None: # test-only
160
+ for point in EXTENSION_POINTS:
161
+ _REGISTRY[point].clear()
162
+ from fi.simulate.simulation import contract as _contract
163
+ _contract._reset_contract_extensions()
@@ -0,0 +1,231 @@
1
+ # ALK harness execution architecture
2
+
3
+ Implementation and test evidence are tracked in
4
+ [`IMPLEMENTATION_AND_VALIDATION_STATUS.md`](IMPLEMENTATION_AND_VALIDATION_STATUS.md). Repository
5
+ packaging behavior and the conformance matrix are documented in
6
+ [`ENVIRONMENT_CONFORMANCE.md`](ENVIRONMENT_CONFORMANCE.md).
7
+
8
+ ## Ownership
9
+
10
+ ALK owns all execution behavior. The Future AGI platform is a control and data plane only.
11
+
12
+ | Concern | ALK package | Platform | Hosted sandbox fleet |
13
+ |---|---:|---:|---:|
14
+ | Understand repository and agent | yes | no | runs ALK |
15
+ | Build databases, tools, mocks and seed data | yes | no | runs ALK |
16
+ | Generate/validate scenarios and personas | yes | no | runs ALK |
17
+ | Connect to and simulate the agent | yes | no | runs ALK |
18
+ | Grade evidence and create artifacts | yes | no | runs ALK |
19
+ | Repository/secret authorization | consumes references | yes | resolves job-scoped values |
20
+ | Job UI, chat, cancellation and history | no | yes | reports status |
21
+ | Event/result/artifact storage | emits data | yes | forwards data |
22
+
23
+ No harness stage imports Temporal, Django, platform models, or platform worker code.
24
+
25
+ ## One engine, two deployments
26
+
27
+ ```text
28
+ Local CLI Hosted product
29
+ ───────── ──────────────
30
+ agent-learn harness auto platform creates HarnessJob
31
+ │ │
32
+ ▼ ▼
33
+ HarnessExecutor isolated ALK sandbox
34
+ │ + HarnessExecutor
35
+ └──────────── same pipeline ─────────────┘
36
+ │
37
+ ▼
38
+ understand → environment → bundle → data → scenarios → connect → simulate → grade
39
+ ```
40
+
41
+ `HarnessJob` is the immutable input boundary. Local jobs use a local repository path. Hosted
42
+ jobs use a GitHub installation/repository reference, archive, image, or remote endpoint and can
43
+ never contain a local path. Agent credentials are `SecretRef` values; resolved secrets are
44
+ rejected from serialized jobs.
45
+
46
+ ## Environment boundary
47
+
48
+ The environment is infrastructure owned by the harness, not a connection to customer
49
+ production. ALK may adopt schemas, migrations, fixtures and mock services from the submitted
50
+ repository. It then fills missing test data and dependencies itself.
51
+
52
+ Every successful build is sealed as an `EnvironmentBundle`:
53
+
54
+ - versioned schema;
55
+ - SHA-256 content address;
56
+ - exact source/generator provenance;
57
+ - runtime document and service list;
58
+ - named capabilities and readiness probes;
59
+ - per-file hashes and sizes;
60
+ - no symlinks or resolved secrets;
61
+ - immutable verification before execution.
62
+
63
+ This removes repository-path assumptions from the runtime. Local Compose and a future hosted
64
+ Kubernetes/Firecracker provider implement the same `RuntimeProvider` interface and consume the
65
+ same manifest.
66
+
67
+ ## Provisioning policy
68
+
69
+ The current local provider follows this order:
70
+
71
+ 1. Detect the submitted Compose definition and declared default infrastructure services.
72
+ 2. Give the run a unique Compose project and free host ports.
73
+ 3. Exclude opt-in agent/worker services from infrastructure startup.
74
+ 4. Build and wait for declared health checks once.
75
+ 5. Derive only the endpoint configuration the agent already reads.
76
+ 6. Reuse a healthy build only when its complete source fingerprint matches.
77
+ 7. If no Compose file exists but a Dockerfile does, generate only supported infrastructure
78
+ declared by the contract (currently Postgres, ClickHouse and Redis) and compose it around the
79
+ submitted runtime. Never generate agent tools or proprietary service behavior.
80
+ 8. Reset between scenarios from a verified snapshot or isolated lifecycle reset.
81
+ 9. Remove the exact project and its test volumes during cleanup.
82
+
83
+ Unknown database engines do not silently fall back to SQLite or Postgres. A store adapter can be
84
+ generated against the engine's native driver, but it must pass generic freeze/restore/counter
85
+ drift and mutation gates before scenarios may use it.
86
+
87
+ ## Evidence and grading invariants
88
+
89
+ - Setup calls are never credited to the agent.
90
+ - A missing agent call cannot satisfy a call-dependent check.
91
+ - Checks must fail against an empty or deliberately damaged world.
92
+ - Dependent checks cannot pass when their prerequisite action never happened.
93
+ - Tool refusal, agent failure, simulator failure, connectivity failure, environment failure,
94
+ infrastructure failure and grading failure remain distinct.
95
+ - Transcripts, semantic calls, resulting state, state diffs and recordings are retained according
96
+ to artifact policy.
97
+ - Agent behavior failures are valid RL results; harness/infrastructure failures are not scored as
98
+ agent failures.
99
+
100
+ ## Repository-backed chat execution
101
+
102
+ Chat and voice share the same autonomous lifecycle. Voice has a standard realtime rendezvous;
103
+ chat runtimes instead declare the conversational ingress they already implement in the grounded
104
+ agent contract. Today a repository runtime can expose either:
105
+
106
+ - HTTP using ALK's turn envelope; or
107
+ - HTTP using an OpenAI Chat Completions-compatible envelope; or
108
+ - a JSON turn exchange over WebSocket.
109
+
110
+ For every scenario ALK binds the restored world, starts the submitted Compose/Dockerfile/generated
111
+ runtime with the environment's private endpoint overrides, exposes the declared container port on
112
+ an ephemeral loopback port locally (or only the private project network in a hosted runner), waits
113
+ for readiness, drives the conversation, records tool effects and removes the runtime. A default
114
+ Compose API is identified by its declared ingress port and excluded from infrastructure startup;
115
+ ambiguous services fail admission rather than being guessed.
116
+
117
+ If the submitted agent returns model-facing tool calls, ALK executes them against the same world
118
+ and continues the turn with tool results. If the agent executes tools itself through the injected
119
+ environment endpoints, the world records those calls at the service boundary. Either way, setup
120
+ activity remains separate from agent evidence.
121
+
122
+ An agent with no external HTTP/WebSocket ingress is not silently reconstructed. Callable/CLI-only
123
+ repository runtimes need a sandbox-side process adapter in a later extension; until then admission
124
+ reports the missing interface explicitly. This preserves the invariant that ALK runs submitted
125
+ agent behavior rather than inventing it.
126
+
127
+ ## Data and scenario quality
128
+
129
+ Scenario validation rejects predictable/demo fixtures such as `123456`, recycled identities,
130
+ and reused payment/booking placeholders. A suite must vary identities, communication styles,
131
+ locations, account/payment states, instructions and expected paths. Submitted seed data is
132
+ preserved where useful and expanded with synthetic records when it is too sparse to exercise the
133
+ contract.
134
+
135
+ The simulator is constrained by literal scenario facts, tracks facts already stated, answers the
136
+ agent's current question, detects rephrased loops, and only retries infrastructure failures.
137
+ Deterministic agent weaknesses remain deterministic failures.
138
+
139
+ ## Delivery and recovery
140
+
141
+ All progress uses ALK's canonical, versioned event envelope. `EventOutbox` writes events and
142
+ fsyncs them before attempting upload. Platform delivery is batchable and idempotent by event ID;
143
+ partial acknowledgements leave the remainder pending. A local run therefore completes offline
144
+ and can sync later. Hosted execution uses the same protocol.
145
+
146
+ The existing Future AGI result sink remains responsible for platform run rows, transcripts,
147
+ evaluations and recording upload. Platform views render stored data; they do not reconstruct or
148
+ run harness stages.
149
+
150
+ ## Scaling and isolation
151
+
152
+ One job maps to one ephemeral hosted sandbox and one resource envelope. The scheduler may place
153
+ those sandboxes on Kubernetes pods or micro-VMs, but that decision is outside ALK. Required
154
+ production controls are:
155
+
156
+ - dedicated execution cluster/account, never ordinary platform workers;
157
+ - per-job filesystem, network namespace and service identity;
158
+ - deny-by-default egress with explicit provider/GitHub/platform destinations;
159
+ - CPU, memory, disk, duration and concurrency quotas from `RuntimeRequirements`;
160
+ - short-lived repository and provider credentials;
161
+ - no privileged containers or host Docker socket inside untrusted sandboxes;
162
+ - artifact size/retention enforcement;
163
+ - cancellation, orphan reconciliation and guaranteed cleanup;
164
+ - cache only content-addressed dependency/image layers, never mutable customer workspaces.
165
+
166
+ ## Extension points
167
+
168
+ - `SourceAcquirer`: GitHub, archive, image or other source materialization.
169
+ - `RuntimeProvider`: local Compose today; isolated hosted provider next.
170
+ - ALK endpoint adapters: callable/local, HTTP, WebSocket, LiveKit, Vapi and Retell today; MCP and
171
+ process/container connectors are the next adapters and must use the same registry.
172
+ - Store registry: Postgres, SQLite and in-process today; generated native adapters for new
173
+ engines after conformance proofs.
174
+ - `EventTransport` and result sinks: local filesystem, Future AGI platform or customer-owned
175
+ telemetry.
176
+
177
+ Adding an environment engine, agent connector, source type or scheduler should be one adapter;
178
+ it must not add a branch to scenario generation or grading.
179
+
180
+ ## Agent/tool ownership boundary
181
+
182
+ ALK provisions the environment around submitted agent code; it never supplies missing agent
183
+ behavior. The submitted repository remains authoritative for prompts, tool schemas, tool
184
+ implementations, orchestration and business rules. Environment adaptation is limited to:
185
+
186
+ - starting declared infrastructure such as databases, queues, object stores and media services;
187
+ - injecting non-secret endpoints through configuration seams the submitted code already reads;
188
+ - resolving referenced credentials at runtime;
189
+ - seeding and resetting test-owned dependency state; and
190
+ - capturing calls, tool evidence and generated artifacts without changing their meaning.
191
+
192
+ A customer-specific API or missing tool implementation is not infrastructure. If its
193
+ implementation is absent, admission fails with an unsupported/missing dependency result. The
194
+ harness must not generate a substitute, proxy invented behavior, or grade against its own
195
+ replacement. A code-execution service follows the same rule: ALK may provide an isolated runtime
196
+ when the agent already declares that dependency, but the customer's tool decides what code to
197
+ execute and how its outputs are used.
198
+
199
+ ## Admission, credentials and source trust
200
+
201
+ Repository admission is a read-only static preflight. It does not import or execute submitted
202
+ code. The scanner ignores dependencies, tests, generated output, symlinks and oversized files;
203
+ recognizes Python, JavaScript, env-template and Compose declarations; and emits only names,
204
+ purposes and statuses. It models provider alternatives explicitly, so one Gemini API key or one
205
+ complete Vertex credential route satisfies model authentication without asking for every option.
206
+
207
+ Public GitHub source is cloned anonymously. Private source carries a GitHub App installation
208
+ reference; the sandbox's credential broker resolves the short-lived token only for clone and
209
+ injects it through Git configuration environment variables, never process arguments or the job.
210
+ The resolved commit is verified when a commit SHA is supplied. Source fingerprinting hashes
211
+ symlink metadata without following links outside the repository, and the supervisor rejects a
212
+ source tree that changes during execution.
213
+
214
+ Non-secret connector configuration and secret references have separate contracts. Inline secret-
215
+ like configuration keys are rejected by the platform. The worker builds a fresh allowlisted
216
+ environment, resolves only the job's `SecretRef` entries, and does not inherit the supervisor's
217
+ model, cloud, repository or customer credentials.
218
+
219
+ ## Retry and ingestion invariants
220
+
221
+ The supervisor retries only structured failures marked retryable in the infrastructure,
222
+ connectivity or platform-sync domains. Each failed worker attempt is archived separately before
223
+ a clean attempt starts. Agent behavior and grading outcomes are never retried to manufacture a
224
+ pass. GitHub clone uses the same bounded exponential-backoff policy.
225
+
226
+ Before terminal success, the artifact directory is sealed with a content-addressed manifest. The
227
+ seal requires a terminal result and non-empty transcript per scenario, validates referenced
228
+ recordings, rejects secret material and unsafe links, enforces the byte budget, and records every
229
+ retained file's SHA-256, media type and size. Platform result payloads and recording uploads carry
230
+ digests. Ingestion locks the call row, accepts identical retries idempotently, rejects conflicting
231
+ evidence, and stores combined, stereo, customer and assistant recordings as distinct artifacts.
@@ -0,0 +1,246 @@
1
+ # The harness: what it builds and how it proves it
2
+
3
+ The reference for the rebuild. Written after the corrections in `_scenario-generation/context/`
4
+ 7.2, 8.1, 9.1, 10.1 and 12, and it supersedes anything in the code that contradicts it.
5
+
6
+ ---
7
+
8
+ ## The model, in one paragraph
9
+
10
+ The **environment step** builds everything that is common to every test of one agent: the world
11
+ its tools act on, the prompt that drives a simulated user if it has one, and the catalogue of
12
+ sub-goals it can be checked against. Every **scenario** is then only a change on that base: what
13
+ it alters after reset, the instruction substituted into the simulator's prompt, and which
14
+ sub-goals must hold. Nothing about a scenario is a template with slots; the harness writes each
15
+ one, and proves it works before keeping it.
16
+
17
+ ```
18
+ environment step ─────────────────────────────► base: world + simulator prompt + sub-goals
19
+ │
20
+ scenario 1 ──► reset → setup → run → check ───────────────┤
21
+ scenario 2 ──► reset → setup → run → check ───────────────┤
22
+ scenario N ──► reset → setup → run → check ───────────────┘
23
+ ```
24
+
25
+ ---
26
+
27
+ ## 1. The environment step
28
+
29
+ > *"You have to first understand what the f\*\*\* this agent is, from that you will create
30
+ > databases, you'll create a snapshot of the databases first."* — Nikhil, 12
31
+
32
+ It produces four things. All of them are written by the harness. None are hardcoded here.
33
+
34
+ ### 1.1 The world
35
+
36
+ Whatever **this** agent needs, and nothing more. For the drive-thru agent that is a database.
37
+ For a browser agent it is a site. For something else it is a filesystem, a queue, a service — the
38
+ harness decides from the contract what has to exist.
39
+
40
+ It **subclasses ALK's `EnvironmentAdapter`**, so the runners that already exist can drive it:
41
+ `reset` publishes the tools and the starting state, `handle_tool_call` executes one call, and the
42
+ state afterwards is what gets graded. Nothing the harness writes should re-implement a runner.
43
+
44
+ It is **frozen once** as a snapshot. Every scenario restores from that snapshot, so a run is
45
+ repeatable and no scenario can inherit another's leftovers.
46
+
47
+ ### 1.2 The simulator prompt — only where the agent is conversational
48
+
49
+ > *"For voice and chat there is a simulator, and the input is an instruction to that simulator
50
+ > rather than an input to the agent under test. Where there is no actor, variability comes from
51
+ > how the environment is designed."* — 10.1 §4
52
+
53
+ The harness writes one prompt for the simulated user of **this** agent, with variables left open.
54
+ Each scenario supplies the values. The prompt is an artifact of the environment step because it
55
+ is the same for every scenario; only the substituted instruction differs.
56
+
57
+ There is no persona field and no persona library.
58
+
59
+ > *"Drop the gimmicky persona characters. Variability comes from real conditions instead: a new
60
+ > versus an existing user, whether a payment method is on file, addresses."* — 8.1
61
+
62
+ For a browser or coding agent there is no simulator at all; the instruction goes to the agent
63
+ directly.
64
+
65
+ ### 1.3 The sub-goal catalogue
66
+
67
+ > *"Defining the sub-goals is our call. The important property is that they are common across
68
+ > scenarios so the results roll up: if a payment step appears in 50 scenarios, the analytics
69
+ > should show where payment fails and how often."* — 10.1 §8
70
+
71
+ Defined **once**, here, as a named list. Scenarios reference them; they do not invent their own
72
+ wording. That is what makes `order-confirmation fails in 7 of 12 scenarios` a sentence anyone can
73
+ say. Each entry carries its own check (see §3).
74
+
75
+ ### 1.4 The gate
76
+
77
+ The environment is not accepted because it looks right. It is exercised: every tool called with a
78
+ valid call, a nonexistent id, and a missing argument; sequences where state has to carry across
79
+ calls. **A refusal is the environment working; a crash is a defect.** It cannot be saved dirty
80
+ (rows left over from building) or unverified.
81
+
82
+ ---
83
+
84
+ ## 2. A scenario is a change on that base, and it owns a folder
85
+
86
+ ```
87
+ name identifier, and the name of its folder
88
+ use_case which branch of the agent's real use cases this belongs to
89
+ setup.py def setup(world), what changes after reset. Code, because what a
90
+ scenario changes is not necessarily the database alone
91
+ ready.py def ready(world), whether the world holds what this scenario presumes
92
+ instruction the task. For a conversational agent this is substituted into the
93
+ simulator prompt; for a browser or coding agent it goes to the agent
94
+ solution the reference trajectory: what a correct agent would do
95
+ checks/*.py one file per deterministic sub-goal, each runnable on its own
96
+ ```
97
+
98
+ The file is the artifact. Code lives in files rather than as strings inside JSON, and every check
99
+ carries a `__main__` block so a person can run it by hand against what a run left behind and get
100
+ the same answer the harness got.
101
+
102
+ Gone from the old shape: `persona`, `opening`, `goal`, and free-text `must` / `must_not` as the
103
+ primary grading. Scenarios are organised **use case → branch**, not by adversarial flavour.
104
+
105
+ > *"A login flow is not one row with happy/edge inside it; it is many rows: login-with-Google,
106
+ > login-with-Microsoft, forgot-password, sign-up-with-email."* — Nikhil, 7.2
107
+
108
+ ---
109
+
110
+ ## 3. Checks: deterministic by default, judge as the fallback
111
+
112
+ > *"When you have `==` or a python script, then I'll call that deterministic."*
113
+ > *"Most likely we can make things deterministic."*
114
+ > *"Deterministic, if possible. And LLM also, obviously."* — Nikhil, 10
115
+
116
+ | | |
117
+ |---|---|
118
+ | **Deterministic** — an assert, an equality, **a python script** | The default |
119
+ | **Non-deterministic** — an LLM judging whether a sub-goal was met | Only where nothing observable settles it |
120
+
121
+ The trap, in his words: *"you are judging by LLM [so it is non-deterministic], but if you want an
122
+ exact output to be 50, then that is deterministic."* An exact fact checked by a judge is **still
123
+ non-deterministic**. What matters is who decides, not how precise the fact is.
124
+
125
+ A check is code the harness writes, and it has two observable things to work from:
126
+
127
+ 1. **the world afterwards** — rows, files, whatever this environment is
128
+ 2. **the recorded tool calls** — that the call happened, *and with the right arguments*
129
+
130
+ That second one answers the question left open in 7.2: a booking made for 10 PM when 11 PM was
131
+ asked for is a failure, and it is deterministic to detect.
132
+
133
+ The judge is left only with what leaves no trace: whether a refusal was explained, whether a price
134
+ was invented, tone.
135
+
136
+ > Warning from the previous run: *"Judge checkpoints, about a third of all checkpoints, are
137
+ > returned as skipped and not graded."* Leaning on the judge does not merely weaken a result — it
138
+ > silently produces holes.
139
+
140
+ ---
141
+
142
+ ## 4. Three gates on every scenario, before it is kept
143
+
144
+ Terminal-bench's oracle run, which is the reason its tasks are known to be solvable.
145
+
146
+ ### Gate 1. Ready
147
+
148
+ ```
149
+ reset → apply setup → run ready ⇒ must HOLD
150
+ ```
151
+
152
+ The world must hold what the scenario presumes. A scenario about the last five items is only a
153
+ test of the agent if there really are five; otherwise the agent fails for a precondition we got
154
+ wrong, and the report reads as a finding about the agent. A missing precondition is ours, and this
155
+ is where it is caught.
156
+
157
+ ### Gate 2. Solvable
158
+
159
+ ```
160
+ reset → apply setup → run the solution → run the checks ⇒ must PASS
161
+ ```
162
+
163
+ If the checks fail with the reference solution, either the scenario is impossible or the check is
164
+ wrong. Both have already happened here: a scenario asserted a value the agent was never permitted
165
+ to send, and another demanded confirmation of an item that could not be ordered. This catches
166
+ them at write time, with no model involved.
167
+
168
+ ### Gate 3. Not vacuous
169
+
170
+ ```
171
+ reset → apply setup → run NOTHING → run the checks ⇒ must FAIL
172
+ ```
173
+
174
+ A check that passes without the agent doing anything grades nothing while reporting a result.
175
+ This is the failure that makes a suite quietly green.
176
+
177
+ Vacuity is judged on *all* checks passing, because one check surviving an empty run ("no
178
+ unavailable item was ordered") is legitimate. A single check that survives is still named, because
179
+ sub-goals are shared: a check that cannot fail without calls would roll up as a pass for an agent
180
+ that did nothing at all.
181
+
182
+ Neither gate asks a model anything. The environment decides.
183
+
184
+ **Three things fall out of the solution for free:** it is the expected trajectory; comparing the
185
+ agent's trajectory against it gives efficiency (Nikhil's point about the agent that succeeds on
186
+ the 21st call after 20 failures); and a scenario that cannot be run is caught before a call is
187
+ ever placed.
188
+
189
+ ---
190
+
191
+ ## 5. Running it
192
+
193
+ > *"Use an existing harness. Just for Claude agents. Use any existing harness that is there."*
194
+ > — Nikhil, 12
195
+
196
+ The simulation runs through **ALK's own path**, not a loop written here. The world is passed in
197
+ as the environment; the agent under test is the real agent, in its real runtime. For the voice
198
+ case that means the Vapi assistant we already have, over LiveKit, with the tool webhook answered
199
+ by **our world** rather than by canned mocks.
200
+
201
+ That last part is the whole point of the environment. The previous run's known issues were:
202
+
203
+ - *"Mocked tools always succeed, including removing an item that was never added."*
204
+ - *"Mock responses do not vary by argument, so read-after-write flows are wrong."*
205
+ - *"World state does not change unless a scenario sets `state_updates`, which is often empty."*
206
+
207
+ A world that really holds rows and can really refuse removes all three.
208
+
209
+ ---
210
+
211
+ ## 6. What changes per kind of agent, and what does not
212
+
213
+ | | Voice / chat | Browser | Coding |
214
+ |---|---|---|---|
215
+ | World | database, KB | a site | a filesystem, a repo |
216
+ | Simulated user | yes — prompt written by the harness | none | none |
217
+ | Instruction goes to | the simulator | the agent | the agent |
218
+ | Solution | tool calls | actions | commands |
219
+ | Check | code over world + calls | code over the page + actions | code over the tree |
220
+
221
+ **What never changes:** the environment is built once and frozen; a scenario is a change on it; a
222
+ solution proves it is solvable; a scenario whose world is not ready is rejected before it can be
223
+ blamed on the agent; a check that cannot fail is rejected; deterministic first.
224
+
225
+ ---
226
+
227
+ ## 7. Order of work
228
+
229
+ 1. **Environment step** — the world, the simulator prompt, the sub-goal catalogue, all written by
230
+ the harness rather than by a fixed schema here.
231
+ 2. **Scenario shape**: a folder per scenario: setup / ready / instruction / solution / sub-goal
232
+ references / a file per check.
233
+ 3. **The three gates**: ready, solvable, and not vacuous.
234
+ 4. **Run through ALK** — the world serving the tool calls of the real agent.
235
+
236
+ ---
237
+
238
+ ## 8. Instructions, not code
239
+
240
+ > *"This is a flow, this is not a harness. You will give your harness the instructions that you
241
+ > are supposed to do all this and then the harness will do all that. It's not a code that your
242
+ > harness follows."* — Nikhil, 12
243
+
244
+ Every stage's method lives in a `SKILL.md`, editable without touching code. What stays in code is
245
+ only what must be exact: executing a call, restoring a snapshot, running a check, and refusing
246
+ something that does not hold up. **The harness decides what to do. Code decides what is true.**