agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,719 @@
1
+ """`futureagi.environment-bundle.v2` — the hosted provisioner's manifest shape (`hosted-execution-
2
+ seams.md` v1.9).
3
+
4
+ v1 (`bundle.py`) describes a `command`-per-service compose world and embeds the repository
5
+ source. v2 describes `/work/source` as already present and a job that starts plain processes on
6
+ localhost: `processes` (managed engines and copied-and-built source trees), `seed` (how each
7
+ store's baseline is built and proven), and the same `capabilities`/`readiness`/`files`/
8
+ `provenance` shape widened for both. v1 stays untouched — this module is additive, not a
9
+ replacement, and the two schema versions are never interchangeable: a hosted provisioner that
10
+ receives a `…bundle.v1` manifest rejects it rather than guessing.
11
+
12
+ What lives here is model-layer only: the shapes, the closed vocabularies, and the rules that need
13
+ nothing but the manifest's own fields to decide. Rules that need the bundle's actual files (secret
14
+ scanning, digest/file verification) or the job it will run under (`compose_not_hosted`,
15
+ `engine_unsupported`, `no_sql_store`, the `depends_on` graph, placeholder-vocabulary checking,
16
+ reserved-name scanning of migration content) are the §2e preflight checklist's job, not this
17
+ module's — see `hosted-execution-seams.md` §2e. Also deferred to that preflight: translating
18
+ pydantic's `extra="forbid"` rejection of an unknown process-entry key into §2b's `unknown_field`
19
+ code — that translation belongs where error surfacing is owned, not here.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import hashlib
25
+ import json
26
+ import re
27
+ from collections import Counter
28
+ from enum import Enum
29
+ from pathlib import Path
30
+ from typing import Annotated, Any, Literal, Sequence, Union
31
+
32
+ from pydantic import (
33
+ BaseModel,
34
+ ConfigDict,
35
+ Field,
36
+ JsonValue,
37
+ ValidationError,
38
+ field_validator,
39
+ model_validator,
40
+ )
41
+
42
+ from .bundle import CapabilityProtocol, _reject_secret_values, _safe_relative
43
+
44
+ BUNDLE_V2_SCHEMA_VERSION = "futureagi.environment-bundle.v2"
45
+ BUNDLE_V2_MANIFEST = "manifest.json"
46
+
47
+
48
+ class BundleV2Error(RuntimeError):
49
+ """A v2 bundle manifest is wrong-versioned, malformed, or fails a model-layer rule."""
50
+
51
+
52
+ # --- §2a runtime -------------------------------------------------------------------------------
53
+
54
+
55
+ class RuntimeKindV2(str, Enum):
56
+ PROCESS = "process"
57
+ EXTERNAL = "external"
58
+ COMPOSE = "compose"
59
+
60
+
61
+ class EvidenceSeam(str, Enum):
62
+ HTTP_TOOL = "http_tool"
63
+ TOOL_TRACE = "tool_trace"
64
+
65
+
66
+ class BundleRuntimeV2(BaseModel):
67
+ model_config = ConfigDict(extra="forbid")
68
+
69
+ kind: RuntimeKindV2
70
+ control_service: str | None = None
71
+ evidence_seam: EvidenceSeam | None = None
72
+ # Carried over from v1 for `kind: compose` only (local SDK runs); a hosted `process` bundle
73
+ # has no document to point at, since `/work/source` is already on disk.
74
+ document: str | None = None
75
+
76
+ @model_validator(mode="after")
77
+ def _kind_specific_rules(self) -> "BundleRuntimeV2":
78
+ if self.kind is RuntimeKindV2.PROCESS and self.evidence_seam is None:
79
+ raise ValueError("evidence_seam_required: kind=process")
80
+ if self.kind is RuntimeKindV2.COMPOSE and not self.document:
81
+ raise ValueError("compose_runtime_requires_document")
82
+ if self.kind is not RuntimeKindV2.COMPOSE and self.document is not None:
83
+ raise ValueError("document_only_for_compose")
84
+ if self.document:
85
+ _safe_relative(self.document)
86
+ return self
87
+
88
+
89
+ # --- §2b processes -------------------------------------------------------------------------
90
+
91
+
92
+ class ProcessKind(str, Enum):
93
+ MANAGED = "managed"
94
+ SOURCE = "source"
95
+
96
+
97
+ class ManagedEngine(str, Enum):
98
+ POSTGRES = "postgres"
99
+ REDIS = "redis"
100
+ RABBITMQ = "rabbitmq"
101
+
102
+
103
+ class ProcessUser(str, Enum):
104
+ """The snapshot's fixed, bundle-assignable users (§0). `svc-control` runs ALK itself and is
105
+ never a process's own user."""
106
+
107
+ SVC_AGENT = "svc-agent"
108
+ SVC_TOOLS = "svc-tools"
109
+ SVC_DATA = "svc-data"
110
+
111
+
112
+ class SecretPurpose(str, Enum):
113
+ TARGET_PROVIDER = "target_provider"
114
+ SIMULATOR_PROVIDER = "simulator_provider"
115
+ SOURCE_CHECKOUT = "source_checkout"
116
+
117
+
118
+ # §0 (v1.8): a process `name` is path-joined into `/work/build/<name>/` and
119
+ # `/work/worlds/w<N>/<name>/` verbatim (§2b) — the pattern below is the closed shape that makes
120
+ # `/`, `..`, and an absolute form unspellable at the model layer, matching every §2b example
121
+ # (including `tools-api`).
122
+ _PROCESS_NAME_PATTERN = re.compile(r"^[a-z0-9][a-z0-9_-]*$")
123
+
124
+
125
+ def _validate_process_name(name: str) -> str:
126
+ if not _PROCESS_NAME_PATTERN.fullmatch(name):
127
+ raise ValueError(
128
+ f"process_name_invalid: {name!r} must match ^[a-z0-9][a-z0-9_-]*$"
129
+ )
130
+ return name
131
+
132
+
133
+ class StartedCheck(BaseModel):
134
+ model_config = ConfigDict(extra="forbid")
135
+
136
+ # §2b (v1.8): "the value selects the port-probe variant, it is not a literal port number" —
137
+ # the probed port is always the dependency's own allocated port (`port_plan.port_for`,
138
+ # `process_runtime.py`), honoring `fixed_port` when the process declares one. A prior version
139
+ # of this field carried a literal int; `bool` makes the "not a literal" rule unspellable
140
+ # wrong rather than merely documented.
141
+ port: bool | None = None
142
+ log_marker: str | None = None
143
+ timeout_seconds: float = Field(default=30.0, gt=0)
144
+
145
+ @model_validator(mode="after")
146
+ def _exactly_one_probe(self) -> "StartedCheck":
147
+ has_port = bool(self.port)
148
+ has_marker = self.log_marker is not None
149
+ if has_port == has_marker:
150
+ raise ValueError("started_check_requires_exactly_one_of_port_or_log_marker")
151
+ return self
152
+
153
+
154
+ class ManagedProcess(BaseModel):
155
+ model_config = ConfigDict(extra="forbid")
156
+
157
+ name: str = Field(min_length=1)
158
+ kind: Literal[ProcessKind.MANAGED] = ProcessKind.MANAGED
159
+ engine: ManagedEngine
160
+ version: str = Field(min_length=1)
161
+ user: ProcessUser
162
+ depends_on: list[str] = Field(default_factory=list)
163
+
164
+ @field_validator("name")
165
+ @classmethod
166
+ def _name_shape(cls, value: str) -> str:
167
+ return _validate_process_name(value)
168
+
169
+
170
+ class SourceProcess(BaseModel):
171
+ model_config = ConfigDict(extra="forbid")
172
+
173
+ name: str = Field(min_length=1)
174
+ kind: Literal[ProcessKind.SOURCE] = ProcessKind.SOURCE
175
+ working_directory: str
176
+ source_origin: Literal["repository", "bundle"] = "repository"
177
+ build_commands: list[list[str]] = Field(default_factory=list)
178
+ run_command: list[str] = Field(min_length=1)
179
+ environment: dict[str, str] = Field(default_factory=dict)
180
+ build_environment: dict[str, str] | None = None
181
+ fixed_port: int | None = Field(default=None, ge=1, le=65535)
182
+ started_check: StartedCheck | None = None
183
+ secret_purposes: list[SecretPurpose] = Field(default_factory=list)
184
+ user: ProcessUser
185
+ depends_on: list[str] = Field(default_factory=list)
186
+
187
+ @field_validator("name")
188
+ @classmethod
189
+ def _name_shape(cls, value: str) -> str:
190
+ return _validate_process_name(value)
191
+
192
+ @model_validator(mode="after")
193
+ def _shape(self) -> "SourceProcess":
194
+ _safe_relative(self.working_directory)
195
+ for step in self.build_commands:
196
+ if not step:
197
+ raise ValueError("build_command_step_empty")
198
+ return self
199
+
200
+
201
+ ProcessEntry = Annotated[
202
+ Union[ManagedProcess, SourceProcess], Field(discriminator="kind")
203
+ ]
204
+
205
+
206
+ # --- §2c seed ------------------------------------------------------------------------------
207
+
208
+
209
+ class BaselineStrategy(str, Enum):
210
+ TEMPLATE_DATABASE = "template_database"
211
+ DATADIR_COPY = "datadir_copy"
212
+ EMPTY = "empty"
213
+
214
+
215
+ # The §2b catalog table. The engine a store answers to is the *capability's* protocol (resolved in
216
+ # the root validator, where `capabilities` is in scope) — a store entry carries no `engine` field
217
+ # of its own, and the sentinel's shape is a proof of that engine, not a second source for it.
218
+ _ENGINE_STRATEGIES: dict[ManagedEngine, frozenset[BaselineStrategy]] = {
219
+ ManagedEngine.POSTGRES: frozenset(
220
+ {BaselineStrategy.TEMPLATE_DATABASE, BaselineStrategy.DATADIR_COPY}
221
+ ),
222
+ ManagedEngine.REDIS: frozenset(
223
+ {BaselineStrategy.DATADIR_COPY, BaselineStrategy.EMPTY}
224
+ ),
225
+ ManagedEngine.RABBITMQ: frozenset({BaselineStrategy.DATADIR_COPY}),
226
+ }
227
+
228
+
229
+ class StoreBaseline(BaseModel):
230
+ model_config = ConfigDict(extra="forbid")
231
+
232
+ strategy: BaselineStrategy
233
+ inputs_digest: str
234
+
235
+ @model_validator(mode="after")
236
+ def _digest_shape(self) -> "StoreBaseline":
237
+ if not re.fullmatch(r"sha256:[0-9a-f]{64}", self.inputs_digest):
238
+ raise ValueError("inputs_digest_invalid")
239
+ return self
240
+
241
+
242
+ class Sentinel(BaseModel):
243
+ """A store's per-protocol read-only proof, per §2c: postgres `{query, expected}`, redis
244
+ `{key, expected}`, rabbitmq `{queue, expected_depth}` — exactly one shape, never a mix."""
245
+
246
+ model_config = ConfigDict(extra="forbid")
247
+
248
+ query: str | None = None
249
+ expected: str | None = None
250
+ key: str | None = None
251
+ queue: str | None = None
252
+ expected_depth: int | None = Field(default=None, ge=0)
253
+
254
+ @model_validator(mode="after")
255
+ def _one_protocol_shape(self) -> "Sentinel":
256
+ if self.implied_engine is None:
257
+ raise ValueError(
258
+ "sentinel_shape_invalid: expected exactly one of "
259
+ "postgres{query,expected}, redis{key,expected}, rabbitmq{queue,expected_depth}"
260
+ )
261
+ return self
262
+
263
+ @property
264
+ def implied_engine(self) -> ManagedEngine | None:
265
+ postgres = self.query is not None and self.expected is not None
266
+ redis = self.key is not None and self.expected is not None
267
+ rabbitmq = self.queue is not None and self.expected_depth is not None
268
+ shapes = [
269
+ (
270
+ postgres,
271
+ ManagedEngine.POSTGRES,
272
+ {self.key, self.queue, self.expected_depth},
273
+ ),
274
+ (redis, ManagedEngine.REDIS, {self.query, self.queue, self.expected_depth}),
275
+ (rabbitmq, ManagedEngine.RABBITMQ, {self.query, self.key, self.expected}),
276
+ ]
277
+ matched = [
278
+ engine for present, engine, others in shapes if present and others == {None}
279
+ ]
280
+ return matched[0] if len(matched) == 1 else None
281
+
282
+
283
+ class StoreEntry(BaseModel):
284
+ model_config = ConfigDict(extra="forbid")
285
+
286
+ capability: str = Field(min_length=1)
287
+ migrations: list[str] = Field(default_factory=list)
288
+ seed_files: list[str] = Field(default_factory=list)
289
+ baseline: StoreBaseline
290
+ sentinel: Sentinel
291
+
292
+ @model_validator(mode="after")
293
+ def _paths(self) -> "StoreEntry":
294
+ for relative_path in (*self.migrations, *self.seed_files):
295
+ _safe_relative(relative_path)
296
+ return self
297
+
298
+
299
+ class Seed(BaseModel):
300
+ model_config = ConfigDict(extra="forbid")
301
+
302
+ stores: list[StoreEntry] = Field(default_factory=list)
303
+
304
+
305
+ # --- §2d capabilities, readiness, files, provenance -----------------------------------------
306
+
307
+
308
+ # §2b's closed placeholder vocabulary, mirrored here (not imported from `process_preflight.py`,
309
+ # which imports this module) so a `configuration_name` can never shadow a builtin token — the
310
+ # reverse dependency direction is preflight -> model, not model -> preflight.
311
+ _RESERVED_CONFIGURATION_NAMES = {"JOB_ID", "WORLD_INDEX", "WORLD_DIR", "DB_NAME"}
312
+ _RESERVED_CONFIGURATION_PREFIX = re.compile(r"^(PORT|HOST)_")
313
+
314
+
315
+ class CapabilityV2(BaseModel):
316
+ model_config = ConfigDict(extra="forbid")
317
+
318
+ protocol: CapabilityProtocol
319
+ service: str = Field(min_length=1)
320
+ container_port: int | None = Field(default=None, ge=1, le=65535)
321
+ configuration_name: str | None = None
322
+
323
+ @model_validator(mode="after")
324
+ def _configuration_name_not_reserved(self) -> "CapabilityV2":
325
+ # A `configuration_name` colliding with a fixed placeholder or a `{{PORT_/HOST_}}` prefix
326
+ # would render the builtin token instead of this capability's address, with no error and
327
+ # no way for the producer to spell the intended value (F8, p4-round1-review).
328
+ name = self.configuration_name
329
+ if name and (
330
+ name in _RESERVED_CONFIGURATION_NAMES
331
+ or _RESERVED_CONFIGURATION_PREFIX.match(name)
332
+ ):
333
+ raise ValueError(f"configuration_name_reserved: {name}")
334
+ return self
335
+
336
+
337
+ class ReadinessProbeV2(BaseModel):
338
+ model_config = ConfigDict(extra="forbid")
339
+
340
+ capability: str
341
+ path: str | None = None
342
+ timeout_seconds: float = Field(default=120.0, gt=0, le=1800)
343
+ interval_seconds: float = Field(default=1.0, gt=0, le=60)
344
+
345
+
346
+ class BundleFileV2(BaseModel):
347
+ model_config = ConfigDict(extra="forbid")
348
+
349
+ path: str
350
+ sha256: str
351
+ size: int = Field(ge=0)
352
+
353
+ @model_validator(mode="after")
354
+ def _valid_path(self) -> "BundleFileV2":
355
+ _safe_relative(self.path)
356
+ if not re.fullmatch(r"[0-9a-f]{64}", self.sha256):
357
+ raise ValueError(f"file_sha256_invalid: {self.path}")
358
+ return self
359
+
360
+
361
+ class BundleProvenanceV2(BaseModel):
362
+ model_config = ConfigDict(extra="forbid")
363
+
364
+ source_kind: str
365
+ repository: str | None = None
366
+ commit: str | None = None
367
+ source_digest: str
368
+ generator: str = "fi.alk.harness"
369
+ generator_version: str = "1"
370
+ adopted_files: list[str] = Field(default_factory=list)
371
+ generated_files: list[str] = Field(default_factory=list)
372
+
373
+ @model_validator(mode="after")
374
+ def _valid_source_digest(self) -> "BundleProvenanceV2":
375
+ # Bare 64-hex, matching what `source_fingerprint` (v1's producer) actually emits — no
376
+ # `sha256:` prefix, unlike `digest`/`inputs_digest`.
377
+ if not re.fullmatch(r"[0-9a-f]{64}", self.source_digest):
378
+ raise ValueError(f"source_digest_invalid: {self.source_digest}")
379
+ return self
380
+
381
+
382
+ # --- manifest root ---------------------------------------------------------------------------
383
+
384
+ _CAPABILITY_SLUG = re.compile(r"[a-z][a-z0-9_]*")
385
+
386
+ # §2c: a store's engine is the engine behind its capability's protocol, not a field the store
387
+ # carries itself. Only postgres/redis/amqp capabilities can host a store at all (§2c); any other
388
+ # protocol on a store's capability is a producer error §2e is left to catch.
389
+ _STORE_ENGINE_BY_PROTOCOL: dict[CapabilityProtocol, ManagedEngine] = {
390
+ CapabilityProtocol.POSTGRES: ManagedEngine.POSTGRES,
391
+ CapabilityProtocol.REDIS: ManagedEngine.REDIS,
392
+ CapabilityProtocol.AMQP: ManagedEngine.RABBITMQ,
393
+ }
394
+
395
+
396
+ class EnvironmentBundleV2(BaseModel):
397
+ model_config = ConfigDict(extra="forbid")
398
+
399
+ schema_version: str
400
+ digest: str
401
+ name: str
402
+ runtime: BundleRuntimeV2
403
+ processes: list[ProcessEntry] = Field(default_factory=list)
404
+ seed: Seed | None = None
405
+ capabilities: dict[str, CapabilityV2] = Field(default_factory=dict)
406
+ readiness: list[ReadinessProbeV2] = Field(default_factory=list)
407
+ files: list[BundleFileV2] = Field(default_factory=list)
408
+ provenance: BundleProvenanceV2
409
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
410
+
411
+ @model_validator(mode="after")
412
+ def _validate_manifest(self) -> "EnvironmentBundleV2":
413
+ if self.schema_version != BUNDLE_V2_SCHEMA_VERSION:
414
+ raise ValueError(f"bundle_schema_unsupported: {self.schema_version}")
415
+ if not re.fullmatch(r"sha256:[0-9a-f]{64}", self.digest):
416
+ raise ValueError("bundle_digest_invalid")
417
+
418
+ if self.runtime.kind is RuntimeKindV2.PROCESS and not self.processes:
419
+ raise ValueError("processes_required: kind=process")
420
+ if self.runtime.kind is RuntimeKindV2.EXTERNAL and (
421
+ self.processes or self.seed is not None
422
+ ):
423
+ raise ValueError("processes_and_seed_forbidden: kind=external")
424
+
425
+ for slug in self.capabilities:
426
+ if not _CAPABILITY_SLUG.fullmatch(slug):
427
+ raise ValueError(f"capability_slug_invalid: {slug}")
428
+
429
+ process_names = Counter(process.name for process in self.processes)
430
+ duplicated_names = sorted(
431
+ name for name, count in process_names.items() if count > 1
432
+ )
433
+ if duplicated_names:
434
+ raise ValueError("process_name_duplicate: " + ", ".join(duplicated_names))
435
+ known_names = set(process_names)
436
+ processes_by_name = {process.name: process for process in self.processes}
437
+
438
+ # B3 (p3-round2-review): only `kind: process` has a `processes` array to resolve against —
439
+ # `external` omits `processes` entirely (§2a) and `compose` addresses services through its
440
+ # own `document`, not this array. Gating here, rather than by emptying `known_names`,
441
+ # keeps the duplicate-name check above meaningful for every runtime kind.
442
+ if self.runtime.kind is RuntimeKindV2.PROCESS:
443
+ service_unresolved = {
444
+ slug: capability.service
445
+ for slug, capability in self.capabilities.items()
446
+ if capability.service not in known_names
447
+ }
448
+ if service_unresolved:
449
+ detail = ", ".join(
450
+ f"{slug}: {service}"
451
+ for slug, service in sorted(service_unresolved.items())
452
+ )
453
+ raise ValueError(f"service_unresolved: {detail}")
454
+
455
+ control_service = self.runtime.control_service
456
+ if control_service is not None and control_service not in known_names:
457
+ raise ValueError(f"control_service_unresolved: {control_service}")
458
+ if control_service is not None and isinstance(
459
+ processes_by_name[control_service], ManagedProcess
460
+ ):
461
+ # §2a: control_service is the agent-side service the world handle and evidence
462
+ # seam attach to — a datastore in that role is incoherent, and would otherwise
463
+ # silently resolve and take svc-agent below (N9, p4-round2-review).
464
+ raise ValueError(
465
+ f"control_service_unresolved: {control_service} is a managed engine, not a "
466
+ "source process"
467
+ )
468
+
469
+ # §2b/§0 (v1.6): the snapshot's SERVICE users are assigned by role, not authored —
470
+ # the control service gets svc-agent, every other source process svc-tools, every
471
+ # managed engine svc-data. Decidable from the manifest's own fields alone once
472
+ # `control_service` is resolved, which is why it lands here rather than in preflight
473
+ # (F5, p4-round1-review).
474
+ for process in self.processes:
475
+ if isinstance(process, ManagedProcess):
476
+ expected_user = ProcessUser.SVC_DATA
477
+ elif process.name == control_service:
478
+ expected_user = ProcessUser.SVC_AGENT
479
+ else:
480
+ expected_user = ProcessUser.SVC_TOOLS
481
+ if process.user is not expected_user:
482
+ raise ValueError(
483
+ f"user_assignment_invalid: {process.name} must be "
484
+ f"{expected_user.value}, got {process.user.value}"
485
+ )
486
+
487
+ names_to_slugs: dict[str, list[str]] = {}
488
+ for slug, capability in self.capabilities.items():
489
+ if capability.configuration_name:
490
+ names_to_slugs.setdefault(capability.configuration_name, []).append(
491
+ slug
492
+ )
493
+ duplicated = {
494
+ name: slugs for name, slugs in names_to_slugs.items() if len(slugs) > 1
495
+ }
496
+ if duplicated:
497
+ detail = ", ".join(
498
+ f"{name} ({', '.join(sorted(slugs))})"
499
+ for name, slugs in sorted(duplicated.items())
500
+ )
501
+ raise ValueError(f"configuration_name_duplicate: {detail}")
502
+
503
+ unresolved = {
504
+ probe.capability
505
+ for probe in self.readiness
506
+ if probe.capability not in self.capabilities
507
+ }
508
+ if self.seed is not None:
509
+ for store in self.seed.stores:
510
+ if store.capability not in self.capabilities:
511
+ unresolved.add(store.capability)
512
+ if unresolved:
513
+ raise ValueError("capability_unresolved: " + ", ".join(sorted(unresolved)))
514
+
515
+ # B1 (p3-round2-review): a capability's *declared* protocol can disagree with the process
516
+ # actually backing it. F19 (p4-round1-review) widened this from "only capabilities with a
517
+ # seed store" to every capability whose protocol names a managed engine — a redis
518
+ # capability with no store entry at all (used only for a `{{...}}` address, never seeded)
519
+ # was previously never checked, and could point `service` at a postgres process silently.
520
+ for slug, capability in self.capabilities.items():
521
+ engine = _STORE_ENGINE_BY_PROTOCOL.get(capability.protocol)
522
+ if engine is None:
523
+ continue
524
+ backing = processes_by_name.get(capability.service)
525
+ if isinstance(backing, ManagedProcess) and backing.engine is not engine:
526
+ raise ValueError(
527
+ f"capability_engine_mismatch: {slug}: protocol {capability.protocol.value} "
528
+ f"resolves to {engine.value}, but {capability.service} is a "
529
+ f"{backing.engine.value} process"
530
+ )
531
+
532
+ if self.seed is not None:
533
+ # Every store's capability resolved above, so its protocol is known — that protocol,
534
+ # not the sentinel's own shape, is the authoritative engine (§2c classifies stores by
535
+ # capability protocol; the sentinel only proves that engine, it doesn't select it).
536
+ for store in self.seed.stores:
537
+ capability = self.capabilities[store.capability]
538
+ engine = _STORE_ENGINE_BY_PROTOCOL.get(capability.protocol)
539
+ if engine is None:
540
+ # B2 (p3-round2-review): a store on a capability outside the three protocols
541
+ # this module knows how to seed (http, mongodb, ...) has no engine to check
542
+ # its sentinel/strategy against — a producer error, not a silent pass-through.
543
+ raise ValueError(
544
+ f"store_protocol_unsupported: {store.capability}: protocol "
545
+ f"{capability.protocol.value} cannot host a seed store"
546
+ )
547
+ backing = processes_by_name.get(capability.service)
548
+ if not isinstance(backing, ManagedProcess):
549
+ # F19 (p4-round1-review): a store on a capability backed by a `SourceProcess`
550
+ # has no managed engine to migrate or seed at all — the all-capabilities
551
+ # engine pass above only fires for a *wrong* managed engine, not a missing one.
552
+ raise ValueError(
553
+ f"store_service_not_managed: {store.capability}: service "
554
+ f"{capability.service!r} is not a managed engine"
555
+ )
556
+ if store.sentinel.implied_engine is not engine:
557
+ raise ValueError(
558
+ f"sentinel_shape_mismatch: {store.capability}: sentinel implies "
559
+ f"{store.sentinel.implied_engine.value}, capability protocol resolves to "
560
+ f"{engine.value}"
561
+ )
562
+ if store.baseline.strategy not in _ENGINE_STRATEGIES[engine]:
563
+ raise ValueError(
564
+ f"seed_strategy_unsupported: {store.capability}: {engine.value} does not "
565
+ f"support {store.baseline.strategy.value}"
566
+ )
567
+
568
+ if self.seed is not None:
569
+ # A store names its capability directly (no placeholder to resolve), so this half of
570
+ # the configuration_name rule is decidable here; the process-`environment` half needs
571
+ # placeholder scanning against the closed `{{...}}` vocabulary, which is §2e's job.
572
+ missing_name = [
573
+ store.capability
574
+ for store in self.seed.stores
575
+ if not self.capabilities[store.capability].configuration_name
576
+ ]
577
+ if missing_name:
578
+ raise ValueError(
579
+ "configuration_name_required: "
580
+ + ", ".join(sorted(set(missing_name)))
581
+ )
582
+
583
+ # v1's whole-manifest resolved-secret guard (`bundle.py`), reapplied here rather than
584
+ # dropped: v2 newly carries free-form `environment`/`build_environment` dicts, exactly
585
+ # where a resolved credential lands if an authoring stage ever inlines one instead of
586
+ # routing it through `secret_purposes`. `secret_purposes` itself is excluded from the
587
+ # dump — its key matches the secret-field pattern (`secret_...`) but it holds purpose
588
+ # identifiers, never values, the same reasoning v1 exempts `secret_refs` under.
589
+ _reject_secret_values(
590
+ self.model_dump(
591
+ exclude={"digest": True, "processes": {"__all__": {"secret_purposes"}}}
592
+ )
593
+ )
594
+ return self
595
+
596
+
597
+ # --- §2c inputs_digest ------------------------------------------------------------------------
598
+
599
+
600
+ def compute_inputs_digest(
601
+ root: str | Path,
602
+ migrations: Sequence[str],
603
+ seed_files: Sequence[str],
604
+ *,
605
+ engine: ManagedEngine,
606
+ version: str,
607
+ ) -> str:
608
+ """The byte-exact `seed.stores[].baseline.inputs_digest` construction from §2c.
609
+
610
+ sha256 over, for each file in ``migrations`` then ``seed_files`` **in listed order** (never
611
+ sorted — order is part of the identity, since migrations must apply in sequence),
612
+ ``<relative_path>\\n<content_length>\\n<content_bytes>``, followed by ``<engine>:<version>\\n``.
613
+ Runs at authoring time, against ``migrations``/``seed_files`` as bundle-relative paths under
614
+ ``root`` (the bundle staging root) — the same paths the sealed manifest records, so the digest
615
+ is reproducible from the bundle's own field values. ``engine``/``version`` must be the store's
616
+ engine's own declared `ManagedProcess.engine`/`.version`, verbatim.
617
+ """
618
+ root = Path(root)
619
+ digest = hashlib.sha256()
620
+ for relative_path in (*migrations, *seed_files):
621
+ _safe_relative(relative_path)
622
+ content = (root / relative_path).read_bytes()
623
+ digest.update(relative_path.encode("utf-8"))
624
+ digest.update(b"\n")
625
+ digest.update(str(len(content)).encode("utf-8"))
626
+ digest.update(b"\n")
627
+ digest.update(content)
628
+ digest.update(f"{engine.value}:{version}\n".encode("utf-8"))
629
+ return "sha256:" + digest.hexdigest()
630
+
631
+
632
+ # --- §2d bundle digest ------------------------------------------------------------------------
633
+
634
+
635
+ def _canonical_json(value: dict[str, Any]) -> bytes:
636
+ return json.dumps(
637
+ value, sort_keys=True, separators=(",", ":"), ensure_ascii=False
638
+ ).encode("utf-8")
639
+
640
+
641
+ def seal_bundle_v2(manifest: EnvironmentBundleV2) -> str:
642
+ """The byte-exact `digest` construction from §2d (v1.7) — the single normative
643
+ implementation; producers call this, never reimplement it.
644
+
645
+ sha256 over the canonical dump of the manifest with ``digest`` and ``files`` removed, then for
646
+ each ``files[]`` record, IN LISTED ORDER, the canonical dump of ``{path, sha256, size}``
647
+ prefixed by its byte length as 8 bytes big-endian. "Canonical" = ``json.dumps(...,
648
+ sort_keys=True, separators=(",", ":"), ensure_ascii=False)``, both times. Operates on
649
+ `BundleFileV2` and `EnvironmentBundleV2.model_dump(mode="json")` directly — v1's `BundleFile`
650
+ never enters this construction, so a field added to v1's model cannot silently rekey a v2
651
+ bundle's digest (F4, p4-round1-review). The hash covers the NORMALIZED dump, so adding an
652
+ optional field to `EnvironmentBundleV2` re-keys every previously sealed bundle — sealer and
653
+ verifier must ship together, which is exactly why there is only one implementation.
654
+ """
655
+ core = manifest.model_dump(mode="json")
656
+ core.pop("digest", None)
657
+ core.pop("files", None)
658
+ digest = hashlib.sha256(_canonical_json(core))
659
+ for record in manifest.files:
660
+ encoded = _canonical_json(record.model_dump(mode="json"))
661
+ digest.update(len(encoded).to_bytes(8, "big"))
662
+ digest.update(encoded)
663
+ return "sha256:" + digest.hexdigest()
664
+
665
+
666
+ def load_bundle_v2(path: str | Path) -> EnvironmentBundleV2:
667
+ """Parse and validate one `futureagi.environment-bundle.v2` manifest.
668
+
669
+ ``path`` is either the manifest file itself or the bundle directory containing it. Schema
670
+ version is checked before full model validation runs, so a `…bundle.v1` manifest — or
671
+ anything else — is named explicitly rather than failing on an unrelated field.
672
+ """
673
+ path = Path(path).expanduser()
674
+ target = path if path.is_file() else path / BUNDLE_V2_MANIFEST
675
+ if not target.is_file():
676
+ raise BundleV2Error(f"bundle_manifest_missing: {target}")
677
+ try:
678
+ raw = json.loads(target.read_text(encoding="utf-8"))
679
+ except (OSError, json.JSONDecodeError) as exc:
680
+ raise BundleV2Error(f"bundle_manifest_invalid: {exc}") from exc
681
+ if not isinstance(raw, dict):
682
+ raise BundleV2Error("bundle_manifest_invalid: not a JSON object")
683
+ schema_version = raw.get("schema_version")
684
+ if schema_version != BUNDLE_V2_SCHEMA_VERSION:
685
+ raise BundleV2Error(f"bundle_schema_unsupported: {schema_version!r}")
686
+ try:
687
+ return EnvironmentBundleV2.model_validate(raw)
688
+ except ValidationError as exc:
689
+ raise BundleV2Error(f"bundle_manifest_invalid: {exc}") from exc
690
+
691
+
692
+ __all__ = [
693
+ "BUNDLE_V2_MANIFEST",
694
+ "BUNDLE_V2_SCHEMA_VERSION",
695
+ "BaselineStrategy",
696
+ "BundleFileV2",
697
+ "BundleProvenanceV2",
698
+ "BundleRuntimeV2",
699
+ "BundleV2Error",
700
+ "CapabilityV2",
701
+ "EnvironmentBundleV2",
702
+ "EvidenceSeam",
703
+ "ManagedEngine",
704
+ "ManagedProcess",
705
+ "ProcessKind",
706
+ "ProcessUser",
707
+ "ReadinessProbeV2",
708
+ "RuntimeKindV2",
709
+ "Seed",
710
+ "SecretPurpose",
711
+ "Sentinel",
712
+ "SourceProcess",
713
+ "StartedCheck",
714
+ "StoreBaseline",
715
+ "StoreEntry",
716
+ "compute_inputs_digest",
717
+ "load_bundle_v2",
718
+ "seal_bundle_v2",
719
+ ]