agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/suite.py ADDED
@@ -0,0 +1,4200 @@
1
+ from __future__ import annotations
2
+
3
+ import asyncio
4
+ import copy
5
+ import hashlib
6
+ import json
7
+ import os
8
+ import shlex
9
+ import sys
10
+ import time
11
+ from dataclasses import dataclass
12
+ from pathlib import Path
13
+ from typing import Any, Mapping, Optional, Sequence
14
+ from xml.sax.saxutils import escape
15
+
16
+ from ._schema import AGENT_LEARNING_CLI_SCHEMA_VERSION, public_payload
17
+
18
+
19
+ AGENT_LEARNING_SUITE_KIND = "agent-learning.suite.v1"
20
+ AGENT_LEARNING_SUITE_OPTIMIZATION_KIND = "agent-learning.suite-optimization.v1"
21
+ AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND = (
22
+ "agent-learning.optimization-lifecycle.v1"
23
+ )
24
+ AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND = (
25
+ "agent-learning.suite.trust-certificate.v1"
26
+ )
27
+ AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND = (
28
+ "agent-learning.suite.trust-verification.v1"
29
+ )
30
+
31
+ _CHILD_COMMANDS = {
32
+ "action_run",
33
+ "baseline",
34
+ "compare",
35
+ "promote_to_regression",
36
+ "replay",
37
+ "report",
38
+ "run",
39
+ "shrink",
40
+ "suite",
41
+ "eval",
42
+ "eval_artifact",
43
+ "eval_task",
44
+ "redteam",
45
+ "optimize",
46
+ "optimize_eval",
47
+ "optimize_suite",
48
+ }
49
+
50
+ _ADMITTED_EVIDENCE_ROLES = {
51
+ "admitted",
52
+ "claim",
53
+ "primary",
54
+ "paper_facing",
55
+ "paper_facing_evidence",
56
+ }
57
+
58
+ _NON_ADMITTED_EVIDENCE_ROLES = {
59
+ "calibration",
60
+ "diagnostic",
61
+ "fixture",
62
+ "preflight",
63
+ "smoke",
64
+ "support",
65
+ }
66
+
67
+
68
+ class SuiteError(ValueError):
69
+ """Raised when an Agent Learning suite manifest cannot run."""
70
+
71
+
72
+ @dataclass(frozen=True)
73
+ class SuiteRunOptions:
74
+ name: Optional[str] = None
75
+ threshold: Optional[float] = None
76
+ max_candidates: Optional[int] = None
77
+ dry_run: bool = False
78
+ fail_fast: bool = False
79
+ require_optimizer_governance: bool = False
80
+
81
+
82
+ @dataclass(frozen=True)
83
+ class SuiteOptimizationOptions:
84
+ name: Optional[str] = None
85
+ threshold: Optional[float] = None
86
+ max_candidates: Optional[int] = None
87
+ dry_run: bool = False
88
+
89
+
90
+ def load_suite_file(path: str | Path) -> dict[str, Any]:
91
+ suite_path = Path(path).expanduser().resolve()
92
+ if not suite_path.exists():
93
+ raise SuiteError(f"suite manifest not found: {suite_path}")
94
+ suite = _load_json_or_yaml(suite_path)
95
+ if not isinstance(suite, Mapping):
96
+ raise SuiteError("suite manifest root must be an object")
97
+ return _prepare_suite(dict(suite), base_dir=suite_path.parent)
98
+
99
+
100
+ def load_suite_artifact_file(path: str | Path) -> dict[str, Any]:
101
+ artifact_path = Path(path).expanduser().resolve()
102
+ if not artifact_path.exists():
103
+ raise SuiteError(f"suite artifact not found: {artifact_path}")
104
+ artifact = _load_json_or_yaml(artifact_path)
105
+ if not isinstance(artifact, Mapping):
106
+ raise SuiteError("suite artifact root must be an object")
107
+ return dict(artifact)
108
+
109
+
110
+ def verify_trust_certificate_file(
111
+ path: str | Path,
112
+ *,
113
+ required_verdict: str = "approved",
114
+ require_promotion_ready: bool = True,
115
+ ) -> dict[str, Any]:
116
+ artifact_path = Path(path).expanduser().resolve()
117
+ artifact = load_suite_artifact_file(artifact_path)
118
+ return verify_trust_certificate(
119
+ artifact,
120
+ required_verdict=required_verdict,
121
+ require_promotion_ready=require_promotion_ready,
122
+ source_path=artifact_path,
123
+ )
124
+
125
+
126
+ def verify_trust_certificate(
127
+ artifact: Mapping[str, Any],
128
+ *,
129
+ required_verdict: str = "approved",
130
+ require_promotion_ready: bool = True,
131
+ source_path: str | Path | None = None,
132
+ ) -> dict[str, Any]:
133
+ """Verify a saved suite trust certificate without re-running the suite."""
134
+ required = _suite_key(required_verdict)
135
+ if required not in _TRUST_VERDICT_RANK:
136
+ allowed = ", ".join(sorted(_TRUST_VERDICT_RANK))
137
+ raise SuiteError(f"required_verdict must be one of: {allowed}")
138
+
139
+ source = Path(source_path).expanduser().resolve() if source_path else None
140
+ result_kind = str(artifact.get("kind") or artifact.get("version") or "")
141
+ summary = _as_mapping(artifact.get("summary"))
142
+ certificate = _as_mapping(artifact.get("trust_certificate"))
143
+ if not certificate and result_kind == AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND:
144
+ certificate = dict(artifact)
145
+
146
+ findings: list[dict[str, Any]] = []
147
+ certificate_kind = str(certificate.get("kind") or "") if certificate else ""
148
+ if not certificate:
149
+ findings.append({
150
+ "type": "suite_trust_certificate_missing",
151
+ "level": "error",
152
+ "reason": "Suite artifact does not contain a trust_certificate block.",
153
+ })
154
+ elif certificate_kind != AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND:
155
+ findings.append({
156
+ "type": "suite_trust_certificate_kind_mismatch",
157
+ "level": "error",
158
+ "reason": (
159
+ "Suite trust certificate kind must be "
160
+ f"{AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND}."
161
+ ),
162
+ "observed_kind": certificate_kind,
163
+ })
164
+
165
+ observed = _suite_key(
166
+ certificate.get("verdict") if certificate else None
167
+ ) or _suite_key(summary.get("trust_certificate_verdict"))
168
+ verdict_rank_passed = False
169
+ if certificate:
170
+ if observed not in _TRUST_VERDICT_RANK:
171
+ findings.append({
172
+ "type": "suite_trust_certificate_verdict_unknown",
173
+ "level": "error",
174
+ "reason": "Suite trust certificate verdict is missing or unknown.",
175
+ "observed_verdict": observed or None,
176
+ })
177
+ else:
178
+ verdict_rank_passed = (
179
+ _TRUST_VERDICT_RANK[observed] >= _TRUST_VERDICT_RANK[required]
180
+ )
181
+ if not verdict_rank_passed:
182
+ findings.append({
183
+ "type": "suite_trust_certificate_verdict_too_low",
184
+ "level": "error",
185
+ "reason": (
186
+ f"Suite trust certificate verdict {observed} is below "
187
+ f"required verdict {required}."
188
+ ),
189
+ "required_verdict": required,
190
+ "observed_verdict": observed,
191
+ })
192
+
193
+ promotion_ready = _optional_bool(
194
+ certificate.get("promotion_ready") if certificate else None,
195
+ summary.get("trust_certificate_promotion_ready"),
196
+ )
197
+ promotion_gate_passed = not require_promotion_ready or promotion_ready is True
198
+ if certificate and not promotion_gate_passed:
199
+ findings.append({
200
+ "type": "suite_trust_certificate_not_promotion_ready",
201
+ "level": "error",
202
+ "reason": "Suite trust certificate is not marked promotion_ready.",
203
+ "promotion_ready": promotion_ready,
204
+ })
205
+
206
+ passed = not findings
207
+ return {
208
+ "kind": AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND,
209
+ "version": AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND,
210
+ "status": "passed" if passed else "failed",
211
+ "exit_code": 0 if passed else 1,
212
+ "source_path": str(source) if source else None,
213
+ "result_kind": result_kind or None,
214
+ "required_verdict": required,
215
+ "require_promotion_ready": bool(require_promotion_ready),
216
+ "observed_verdict": observed or None,
217
+ "promotion_ready": promotion_ready,
218
+ "certificate_kind": certificate_kind or None,
219
+ "assurance_level": (
220
+ certificate.get("assurance_level") if certificate else None
221
+ ),
222
+ "summary": {
223
+ "certificate_present": bool(certificate),
224
+ "certificate_kind_passed": (
225
+ certificate_kind == AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND
226
+ ),
227
+ "verdict_rank_passed": verdict_rank_passed,
228
+ "promotion_gate_passed": promotion_gate_passed,
229
+ "finding_count": len(findings),
230
+ },
231
+ "trust_certificate": copy.deepcopy(certificate),
232
+ "findings": findings,
233
+ }
234
+
235
+
236
+ load_suite = load_suite_file
237
+
238
+
239
+ def required_suite_env(
240
+ suite: Mapping[str, Any],
241
+ *,
242
+ suite_path: str | Path = ".",
243
+ ) -> list[str]:
244
+ base_dir = _suite_base_dir(suite_path)
245
+ required = set(_as_string_list(suite.get("required_env")))
246
+ for job in _suite_jobs(suite):
247
+ try:
248
+ child = _load_child_source(job, base_dir=base_dir)
249
+ except Exception:
250
+ continue
251
+ if _normalize_command(job.get("command") or job.get("type")) == "suite":
252
+ required.update(
253
+ required_suite_env(
254
+ child,
255
+ suite_path=_job_path(job, base_dir=base_dir),
256
+ )
257
+ )
258
+ continue
259
+ required.update(_as_string_list(child.get("required_env")))
260
+ return sorted(required)
261
+
262
+
263
+ def missing_suite_env(
264
+ suite: Mapping[str, Any],
265
+ *,
266
+ suite_path: str | Path = ".",
267
+ ) -> list[str]:
268
+ return [
269
+ key
270
+ for key in required_suite_env(suite, suite_path=suite_path)
271
+ if not os.environ.get(key)
272
+ ]
273
+
274
+
275
+ def validate_suite_env(
276
+ suite: Mapping[str, Any],
277
+ *,
278
+ suite_path: str | Path = ".",
279
+ ) -> None:
280
+ missing = missing_suite_env(suite, suite_path=suite_path)
281
+ if missing:
282
+ raise SuiteError(
283
+ "missing required environment variable(s): "
284
+ f"{', '.join(sorted(missing))}"
285
+ )
286
+
287
+
288
+ def build_suite_manifest(
289
+ *,
290
+ name: str,
291
+ jobs: Sequence[Mapping[str, Any]],
292
+ required_env: Sequence[str] = (),
293
+ required_capabilities: Optional[Mapping[str, Sequence[str]]] = None,
294
+ outputs: Optional[Mapping[str, Any]] = None,
295
+ metadata: Optional[Mapping[str, Any]] = None,
296
+ optimizer_governance_policy: Optional[Mapping[str, Any]] = None,
297
+ threshold: Optional[float] = None,
298
+ fail_fast: Optional[bool] = None,
299
+ ) -> dict[str, Any]:
300
+ """Build an Agent Learning suite manifest from SDK data.
301
+
302
+ This is the SDK counterpart to writing ``agent-learning.suite.v1`` JSON by
303
+ hand: users can compose run/eval/red-team/optimization jobs in Python and
304
+ execute them through ``run_suite`` or ``run_suite_file``.
305
+ """
306
+
307
+ if not name:
308
+ raise ValueError("name is required")
309
+ if not jobs:
310
+ raise ValueError("jobs must contain at least one suite job")
311
+ manifest: dict[str, Any] = {
312
+ "version": AGENT_LEARNING_SUITE_KIND,
313
+ "name": str(name),
314
+ "required_env": _unique_strings(required_env),
315
+ "jobs": [
316
+ _normalize_suite_job(job, index)
317
+ for index, job in enumerate(jobs, start=1)
318
+ ],
319
+ }
320
+ if required_capabilities:
321
+ manifest["required_capabilities"] = {
322
+ str(key): _unique_strings(value)
323
+ for key, value in dict(required_capabilities).items()
324
+ if _unique_strings(value)
325
+ }
326
+ if outputs:
327
+ manifest["outputs"] = copy.deepcopy(dict(outputs))
328
+ if metadata:
329
+ manifest["metadata"] = copy.deepcopy(dict(metadata))
330
+ if optimizer_governance_policy:
331
+ manifest["optimizer_governance_policy"] = copy.deepcopy(
332
+ dict(optimizer_governance_policy)
333
+ )
334
+ if threshold is not None:
335
+ manifest["threshold"] = float(threshold)
336
+ if fail_fast is not None:
337
+ manifest["fail_fast"] = bool(fail_fast)
338
+ return manifest
339
+
340
+
341
+ def build_trinity_suite_manifest(
342
+ *,
343
+ name: str,
344
+ run_path: str | Path,
345
+ eval_path: str | Path,
346
+ artifact_eval_path: str | Path,
347
+ artifact_report_path: str | Path,
348
+ redteam_path: str | Path,
349
+ eval_optimization_path: str | Path,
350
+ optimization_path: str | Path,
351
+ world_model_optimization_path: str | Path | None = None,
352
+ artifact_action_id: str | None = "report_orchestration_strategy",
353
+ artifact_action_cwd: str | Path | None = "artifacts/action-loop/workspace",
354
+ artifact_optimization_path: str | Path | None = None,
355
+ artifact_eval_config_path: str | Path | None = None,
356
+ required_env: Sequence[str] = (),
357
+ max_candidates: Optional[int] = None,
358
+ metadata: Optional[Mapping[str, Any]] = None,
359
+ ) -> dict[str, Any]:
360
+ """Build a run/eval/artifact/red-team/optimization suite.
361
+
362
+ The manifest mirrors the promptfoo-style trinity workflow: simulation,
363
+ text eval, saved-artifact eval, direct artifact-report eval, optional
364
+ artifact-evidence optimization, red-team, eval-suite optimization, and
365
+ full manifest optimization in one capability-gated suite.
366
+ """
367
+
368
+ suite_name = str(name)
369
+ jobs: list[dict[str, Any]] = [
370
+ {
371
+ "id": "local-simulation",
372
+ "command": "run",
373
+ "path": _suite_path_text(run_path),
374
+ "name": f"{suite_name}-run",
375
+ },
376
+ {
377
+ "id": "promptfoo-style-eval",
378
+ "command": "eval",
379
+ "path": _suite_path_text(eval_path),
380
+ "name": f"{suite_name}-eval",
381
+ },
382
+ {
383
+ "id": "artifact-task-eval",
384
+ "command": "eval",
385
+ "path": _suite_path_text(artifact_eval_path),
386
+ "name": f"{suite_name}-artifact-eval",
387
+ },
388
+ {
389
+ "id": "direct-artifact-report-eval",
390
+ "command": "eval-artifact",
391
+ "path": _suite_path_text(artifact_report_path),
392
+ "name": f"{suite_name}-direct-artifact",
393
+ },
394
+ ]
395
+ if artifact_action_id:
396
+ action_job = {
397
+ "id": "artifact-action-report",
398
+ "command": "action-run",
399
+ "path": _suite_path_text(artifact_report_path),
400
+ "action_id": str(artifact_action_id),
401
+ "name": f"{suite_name}-artifact-action-report",
402
+ "output": "../../artifacts/action-loop/action-run.json",
403
+ "outputs": {
404
+ "junit": "../../artifacts/action-loop/action-run.junit.xml",
405
+ "sarif": "../../artifacts/action-loop/action-run.sarif.json",
406
+ "markdown": "../../artifacts/action-loop/action-run.md",
407
+ },
408
+ }
409
+ if artifact_action_cwd is not None:
410
+ action_job["cwd"] = _suite_path_text(artifact_action_cwd)
411
+ jobs.append(action_job)
412
+ if artifact_optimization_path is not None:
413
+ jobs.append(
414
+ {
415
+ "id": "artifact-evidence-optimizer",
416
+ "command": "optimize-eval",
417
+ "path": _suite_path_text(artifact_optimization_path),
418
+ "name": f"{suite_name}-artifact-optimizer",
419
+ }
420
+ )
421
+ jobs.extend(
422
+ [
423
+ {
424
+ "id": "agent-red-team",
425
+ "command": "redteam",
426
+ "path": _suite_path_text(redteam_path),
427
+ "name": f"{suite_name}-redteam",
428
+ },
429
+ {
430
+ "id": "eval-suite-optimizer",
431
+ "command": "optimize-eval",
432
+ "path": _suite_path_text(eval_optimization_path),
433
+ "name": f"{suite_name}-eval-optimizer",
434
+ },
435
+ {
436
+ "id": "agent-optimizer",
437
+ "command": "optimize",
438
+ "path": _suite_path_text(optimization_path),
439
+ "name": f"{suite_name}-optimizer",
440
+ },
441
+ ]
442
+ )
443
+ required_metrics = ["eval_assertions"]
444
+ if world_model_optimization_path is not None:
445
+ jobs.append(
446
+ {
447
+ "id": "world-model-optimizer",
448
+ "command": "optimize",
449
+ "path": _suite_path_text(world_model_optimization_path),
450
+ "name": f"{suite_name}-world-model-optimizer",
451
+ }
452
+ )
453
+ required_metrics.extend(
454
+ [
455
+ "world_contract_quality",
456
+ "world_contract_coverage",
457
+ "tool_selection_accuracy",
458
+ ]
459
+ )
460
+ if artifact_eval_config_path is not None:
461
+ jobs[3]["config"] = _suite_path_text(artifact_eval_config_path)
462
+ if max_candidates is not None:
463
+ for job in jobs:
464
+ if job["command"] in {"optimize", "optimize-eval"}:
465
+ job["max_candidates"] = int(max_candidates)
466
+ return build_suite_manifest(
467
+ name=suite_name,
468
+ required_env=required_env,
469
+ jobs=jobs,
470
+ required_capabilities={
471
+ "commands": [
472
+ "run",
473
+ "eval",
474
+ "eval_artifact",
475
+ "action_run",
476
+ "redteam",
477
+ "optimize_eval",
478
+ "optimize",
479
+ ],
480
+ "result_kinds": [
481
+ "agent-learning.run.v1",
482
+ "agent-learning.eval.v1",
483
+ "agent-learning.artifact-evaluation.v1",
484
+ "agent-learning.action-run.v1",
485
+ "agent-learning.redteam.v1",
486
+ "agent-learning.eval-optimization.v1",
487
+ "agent-learning.optimization.v1",
488
+ ],
489
+ "metrics": required_metrics,
490
+ },
491
+ metadata={
492
+ "source": "fi.alk.suite.build_trinity_suite_manifest",
493
+ **copy.deepcopy(dict(metadata or {})),
494
+ },
495
+ optimizer_governance_policy={
496
+ "require_optimizer_governance": True,
497
+ "min_governed": 1,
498
+ },
499
+ )
500
+
501
+
502
+ def build_framework_adapter_trinity_suite_manifest(
503
+ *,
504
+ name: str,
505
+ run_path: str | Path,
506
+ redteam_path: str | Path,
507
+ required_env: Sequence[str] = (),
508
+ required_frameworks: Sequence[str] = (),
509
+ metadata: Optional[Mapping[str, Any]] = None,
510
+ outputs: Optional[Mapping[str, Any]] = None,
511
+ threshold: Optional[float] = None,
512
+ fail_fast: bool = True,
513
+ ) -> dict[str, Any]:
514
+ """Build a focused suite for framework simulation, eval, and red-team gates."""
515
+
516
+ if not name:
517
+ raise ValueError("name is required")
518
+ suite_name = str(name)
519
+ frameworks = _unique_strings(required_frameworks)
520
+ required_capabilities: dict[str, list[str]] = {
521
+ "commands": ["run", "redteam"],
522
+ "result_kinds": [
523
+ "agent-learning.run.v1",
524
+ "agent-learning.redteam.v1",
525
+ ],
526
+ "metrics": [
527
+ "framework_runtime_contract",
528
+ "framework_adapter_contract_quality",
529
+ "adversarial_resilience",
530
+ "red_team_campaign_quality",
531
+ ],
532
+ }
533
+ if frameworks:
534
+ required_capabilities["frameworks"] = frameworks
535
+ return build_suite_manifest(
536
+ name=suite_name,
537
+ required_env=required_env,
538
+ jobs=[
539
+ {
540
+ "id": "optimized-framework-run",
541
+ "command": "run",
542
+ "path": _suite_path_text(run_path),
543
+ "name": f"{suite_name}-run",
544
+ },
545
+ {
546
+ "id": "framework-red-team",
547
+ "command": "redteam",
548
+ "path": _suite_path_text(redteam_path),
549
+ "name": f"{suite_name}-redteam",
550
+ },
551
+ ],
552
+ required_capabilities=required_capabilities,
553
+ outputs=outputs,
554
+ threshold=threshold,
555
+ fail_fast=fail_fast,
556
+ metadata={
557
+ "source": "fi.alk.suite.build_framework_adapter_trinity_suite_manifest",
558
+ "task_kind": "framework_adapter_trinity_suite",
559
+ **copy.deepcopy(dict(metadata or {})),
560
+ },
561
+ )
562
+
563
+
564
+ def write_framework_adapter_trinity_suite_workspace(
565
+ *,
566
+ name: str,
567
+ framework: str,
568
+ target: str,
569
+ directory: str | Path,
570
+ adapter_candidates: Optional[Sequence[Mapping[str, Any]]] = None,
571
+ agent: Any = None,
572
+ agent_factory: Any = None,
573
+ cases: Sequence[Mapping[str, Any]] = (),
574
+ target_base_dir: str | Path = ".",
575
+ target_factory: Optional[bool] = None,
576
+ method_candidates: Optional[Sequence[str | None]] = None,
577
+ input_mode_candidates: Optional[Sequence[str]] = None,
578
+ required_env: Sequence[str] = (),
579
+ scenario: Optional[Mapping[str, Any]] = None,
580
+ framework_trace: Optional[Mapping[str, Any]] = None,
581
+ evaluation_config: Optional[Mapping[str, Any]] = None,
582
+ auto_evaluation_config: bool = True,
583
+ threshold: float = 0.9,
584
+ trace_runtime: bool = True,
585
+ allow_external_target: bool = False,
586
+ metadata: Optional[Mapping[str, Any]] = None,
587
+ discovery_max_candidates: Optional[int] = 8,
588
+ max_candidates: Optional[int] = None,
589
+ include_seed: bool = True,
590
+ factory: Optional[bool] = None,
591
+ min_turns: int = 1,
592
+ max_turns: int = 1,
593
+ redteam_attacks: Sequence[str] = ("prompt_injection", "credential_exfiltration"),
594
+ redteam_surfaces: Sequence[str] = ("instruction", "tool"),
595
+ redteam_taxonomies: Sequence[str] = ("owasp_llm_top_10", "owasp_agentic_ai"),
596
+ redteam_channels: Sequence[str] = ("chat",),
597
+ redteam_providers: Sequence[str] = ("local_cli",),
598
+ redteam_agent: Optional[Mapping[str, Any]] = None,
599
+ redteam_config: Optional[Mapping[str, Any]] = None,
600
+ redteam_overrides: Optional[Mapping[str, Any]] = None,
601
+ canaries: Sequence[Any] = (),
602
+ blocked_tools: Sequence[str] = (),
603
+ redteam_min_turns: int = 3,
604
+ redteam_max_turns: int = 3,
605
+ ) -> dict[str, Any]:
606
+ """Write a runnable framework adapter run+red-team suite workspace."""
607
+
608
+ if not name:
609
+ raise ValueError("name is required")
610
+ if not framework:
611
+ raise ValueError("framework is required")
612
+ if not target:
613
+ raise ValueError("target is required")
614
+
615
+ workspace = Path(directory).expanduser().resolve()
616
+ manifests_dir = workspace / "manifests"
617
+ manifests_dir.mkdir(parents=True, exist_ok=True)
618
+ selected_target = _suite_local_target_text(target, base_dir=target_base_dir)
619
+ suite_metadata = copy.deepcopy(dict(metadata or {}))
620
+
621
+ from fi.alk import optimize, redteam
622
+
623
+ run_manifest = optimize.build_framework_run_manifest_from_local_adapter(
624
+ name=f"{name}-run",
625
+ framework=framework,
626
+ target=selected_target,
627
+ adapter_candidates=adapter_candidates,
628
+ agent=agent,
629
+ agent_factory=agent_factory,
630
+ cases=cases,
631
+ target_base_dir=target_base_dir,
632
+ target_factory=target_factory,
633
+ method_candidates=method_candidates,
634
+ input_mode_candidates=input_mode_candidates,
635
+ required_env=required_env,
636
+ scenario=scenario,
637
+ framework_trace=framework_trace,
638
+ evaluation_config=evaluation_config,
639
+ auto_evaluation_config=auto_evaluation_config,
640
+ threshold=threshold,
641
+ trace_runtime=trace_runtime,
642
+ allow_external_target=allow_external_target,
643
+ metadata={
644
+ "suite": name,
645
+ "suite_role": "optimized_framework_run",
646
+ **suite_metadata,
647
+ },
648
+ discovery_max_candidates=discovery_max_candidates,
649
+ max_candidates=max_candidates,
650
+ include_seed=include_seed,
651
+ factory=factory,
652
+ min_turns=min_turns,
653
+ max_turns=max_turns,
654
+ )
655
+ run_path = _write_suite_json(
656
+ run_manifest,
657
+ manifests_dir / "optimized-framework-run.json",
658
+ )
659
+
660
+ agent_config = copy.deepcopy(dict(run_manifest.get("agent") or {}))
661
+ agent_metadata = copy.deepcopy(dict(agent_config.get("metadata") or {}))
662
+ redteam_manifest = redteam.build_redteam_manifest(
663
+ name=f"{name}-redteam",
664
+ attacks=redteam_attacks,
665
+ surfaces=redteam_surfaces,
666
+ taxonomies=redteam_taxonomies,
667
+ channels=redteam_channels,
668
+ providers=redteam_providers,
669
+ frameworks=[framework],
670
+ required_env=required_env,
671
+ target={
672
+ "agent": run_manifest.get("name") or f"{name}-run",
673
+ "framework": framework,
674
+ "adapter_target": selected_target,
675
+ "framework_adapter_contract": copy.deepcopy(
676
+ agent_metadata.get("framework_adapter_probe_contract")
677
+ or agent_metadata.get("framework_adapter_contract")
678
+ ),
679
+ "framework_adapter_probe_proof_status": (
680
+ copy.deepcopy(
681
+ dict(agent_metadata.get("framework_adapter_probe_proof") or {})
682
+ ).get("status")
683
+ ),
684
+ "framework_adapter_discovery_used": bool(
685
+ agent_metadata.get("framework_adapter_discovery_used")
686
+ ),
687
+ "suite": name,
688
+ },
689
+ agent=redteam_agent,
690
+ redteam=redteam_overrides,
691
+ evaluation_config=redteam_config,
692
+ threshold=threshold,
693
+ canaries=canaries,
694
+ blocked_tools=blocked_tools,
695
+ min_turns=redteam_min_turns,
696
+ max_turns=redteam_max_turns,
697
+ )
698
+ redteam_path = _write_suite_json(
699
+ redteam_manifest,
700
+ manifests_dir / "framework-redteam.json",
701
+ )
702
+
703
+ suite_manifest = build_framework_adapter_trinity_suite_manifest(
704
+ name=name,
705
+ run_path=Path("manifests") / run_path.name,
706
+ redteam_path=Path("manifests") / redteam_path.name,
707
+ required_env=required_env,
708
+ required_frameworks=[framework],
709
+ threshold=threshold,
710
+ metadata={
711
+ "source": "fi.alk.suite.write_framework_adapter_trinity_suite_workspace",
712
+ "framework": framework,
713
+ "target": selected_target,
714
+ **suite_metadata,
715
+ },
716
+ )
717
+ suite_path = write_suite_file(suite_manifest, workspace / "suite.json")
718
+ return {
719
+ "kind": "agent-learning.framework-adapter-trinity-workspace.v1",
720
+ "status": "passed",
721
+ "name": str(name),
722
+ "summary": {
723
+ "framework": framework,
724
+ "target": selected_target,
725
+ "suite_job_count": len(suite_manifest["jobs"]),
726
+ "run_manifest": str(run_path),
727
+ "redteam_manifest": str(redteam_path),
728
+ "suite_manifest": str(suite_path),
729
+ },
730
+ "paths": {
731
+ "workspace": str(workspace),
732
+ "suite": str(suite_path),
733
+ "run": str(run_path),
734
+ "redteam": str(redteam_path),
735
+ },
736
+ "suite": suite_manifest,
737
+ "run_manifest": run_manifest,
738
+ "redteam_manifest": redteam_manifest,
739
+ }
740
+
741
+
742
+ def build_framework_adapter_trinity_suite_optimization_manifest(
743
+ *,
744
+ name: str,
745
+ run_path: str | Path,
746
+ trinity_suite_path: str | Path,
747
+ framework: str,
748
+ required_env: Sequence[str] = (),
749
+ required_frameworks: Sequence[str] = (),
750
+ metadata: Optional[Mapping[str, Any]] = None,
751
+ threshold: float = 1.0,
752
+ optimizer: Optional[Mapping[str, Any]] = None,
753
+ ) -> dict[str, Any]:
754
+ """Build a suite optimization that selects full framework trinity coverage."""
755
+
756
+ if not name:
757
+ raise ValueError("name is required")
758
+ if not framework:
759
+ raise ValueError("framework is required")
760
+ suite_name = str(name)
761
+ frameworks = _unique_strings(required_frameworks) or [str(framework)]
762
+ seed_job = {
763
+ "id": "optimized-framework-run",
764
+ "command": "run",
765
+ "path": _suite_path_text(run_path),
766
+ "name": f"{suite_name}-run-only-seed",
767
+ }
768
+ trinity_job = {
769
+ "id": "framework-adapter-trinity",
770
+ "command": "suite",
771
+ "path": _suite_path_text(trinity_suite_path),
772
+ "name": f"{suite_name}-full-trinity",
773
+ }
774
+ manifest = build_suite_manifest(
775
+ name=suite_name,
776
+ required_env=required_env,
777
+ jobs=[seed_job],
778
+ required_capabilities={
779
+ "commands": ["run", "redteam", "suite"],
780
+ "result_kinds": [
781
+ "agent-learning.run.v1",
782
+ "agent-learning.redteam.v1",
783
+ "agent-learning.suite.v1",
784
+ ],
785
+ "frameworks": frameworks,
786
+ "metrics": [
787
+ "framework_runtime_contract",
788
+ "framework_adapter_contract_quality",
789
+ "adversarial_resilience",
790
+ "red_team_campaign_quality",
791
+ ],
792
+ },
793
+ metadata={
794
+ "source": (
795
+ "fi.alk.suite."
796
+ "build_framework_adapter_trinity_suite_optimization_manifest"
797
+ ),
798
+ "task_kind": "framework_adapter_trinity_suite_optimization",
799
+ "framework": framework,
800
+ **copy.deepcopy(dict(metadata or {})),
801
+ },
802
+ )
803
+ manifest["optimization"] = {
804
+ "threshold": float(threshold),
805
+ "target": {
806
+ "name": suite_name,
807
+ "layers": ["harness", "framework", "security", "evaluator"],
808
+ "base_config": {"jobs": [copy.deepcopy(seed_job)]},
809
+ "search_space": {
810
+ "jobs.0": [
811
+ copy.deepcopy(seed_job),
812
+ copy.deepcopy(trinity_job),
813
+ ]
814
+ },
815
+ "metadata": {
816
+ "source": (
817
+ "fi.alk.suite."
818
+ "build_framework_adapter_trinity_suite_optimization_manifest"
819
+ ),
820
+ "task_kind": "framework_adapter_trinity_suite_optimization",
821
+ "framework": framework,
822
+ **copy.deepcopy(dict(metadata or {})),
823
+ },
824
+ },
825
+ "optimizer": copy.deepcopy(
826
+ dict(
827
+ optimizer
828
+ or {
829
+ "algorithm": "agent",
830
+ "max_candidates": 3,
831
+ "include_seed": True,
832
+ "auto_diagnose": False,
833
+ }
834
+ )
835
+ ),
836
+ }
837
+ return manifest
838
+
839
+
840
+ def write_framework_adapter_trinity_suite_optimization_workspace(
841
+ *,
842
+ name: str,
843
+ framework: str,
844
+ target: str,
845
+ directory: str | Path,
846
+ suite_optimization_threshold: float = 1.0,
847
+ suite_optimizer: Optional[Mapping[str, Any]] = None,
848
+ **workspace_kwargs: Any,
849
+ ) -> dict[str, Any]:
850
+ """Write a framework trinity workspace plus an optimizable outer suite."""
851
+
852
+ workspace = write_framework_adapter_trinity_suite_workspace(
853
+ name=name,
854
+ framework=framework,
855
+ target=target,
856
+ directory=directory,
857
+ **workspace_kwargs,
858
+ )
859
+ workspace_root = Path(workspace["paths"]["workspace"]).expanduser().resolve()
860
+ metadata = copy.deepcopy(dict(workspace_kwargs.get("metadata") or {}))
861
+ optimization_manifest = build_framework_adapter_trinity_suite_optimization_manifest(
862
+ name=f"{name}-optimization",
863
+ run_path=Path("manifests") / Path(workspace["paths"]["run"]).name,
864
+ trinity_suite_path=Path("suite.json"),
865
+ framework=framework,
866
+ required_env=workspace_kwargs.get("required_env", ()),
867
+ required_frameworks=[framework],
868
+ metadata={
869
+ "source": (
870
+ "fi.alk.suite."
871
+ "write_framework_adapter_trinity_suite_optimization_workspace"
872
+ ),
873
+ "framework": framework,
874
+ "target": workspace["summary"]["target"],
875
+ **metadata,
876
+ },
877
+ threshold=suite_optimization_threshold,
878
+ optimizer=suite_optimizer,
879
+ )
880
+ optimization_path = write_suite_file(
881
+ optimization_manifest,
882
+ workspace_root / "suite-optimization.json",
883
+ )
884
+ return {
885
+ "kind": "agent-learning.framework-adapter-trinity-optimization-workspace.v1",
886
+ "status": "passed",
887
+ "name": str(name),
888
+ "summary": {
889
+ **copy.deepcopy(dict(workspace.get("summary") or {})),
890
+ "suite_optimization_manifest": str(optimization_path),
891
+ "suite_optimization_search_paths": ["jobs.0"],
892
+ },
893
+ "paths": {
894
+ **copy.deepcopy(dict(workspace.get("paths") or {})),
895
+ "suite_optimization": str(optimization_path),
896
+ },
897
+ "suite_optimization": optimization_manifest,
898
+ "trinity_workspace": workspace,
899
+ }
900
+
901
+
902
+ def build_regression_artifact_suite_manifest(
903
+ *,
904
+ name: str,
905
+ baseline_path: str | Path,
906
+ current_path: str | Path,
907
+ finding_path: str | Path,
908
+ replay_manifest_paths: Sequence[str | Path],
909
+ required_env: Sequence[str] = (),
910
+ min_score_delta: float = 0.0,
911
+ max_new_findings: int = 0,
912
+ max_new_error_findings: int = 0,
913
+ min_level: str = "warning",
914
+ max_findings: int = 1,
915
+ metadata: Optional[Mapping[str, Any]] = None,
916
+ ) -> dict[str, Any]:
917
+ """Build the artifact-regression lifecycle suite from SDK paths.
918
+
919
+ This composes the lifecycle users usually script around CI artifacts:
920
+ create a compact baseline, compare current vs baseline, render a report,
921
+ promote a red-team finding into a regression manifest, and replay one or
922
+ more regression manifests.
923
+ """
924
+
925
+ replay_paths = [_suite_path_text(path) for path in replay_manifest_paths]
926
+ if not replay_paths:
927
+ raise ValueError("replay_manifest_paths must contain at least one manifest")
928
+
929
+ suite_name = str(name)
930
+ jobs = [
931
+ {
932
+ "id": "baseline-current-run",
933
+ "command": "baseline",
934
+ "path": _suite_path_text(current_path),
935
+ "name": f"{suite_name}-baseline",
936
+ },
937
+ {
938
+ "id": "compare-baseline-to-current",
939
+ "command": "compare",
940
+ "path": _suite_path_text(current_path),
941
+ "baseline": _suite_path_text(baseline_path),
942
+ "current": _suite_path_text(current_path),
943
+ "name": f"{suite_name}-compare",
944
+ "min_score_delta": float(min_score_delta),
945
+ "max_new_findings": int(max_new_findings),
946
+ "max_new_error_findings": int(max_new_error_findings),
947
+ },
948
+ {
949
+ "id": "report-current-run",
950
+ "command": "report",
951
+ "path": _suite_path_text(current_path),
952
+ "name": f"{suite_name}-report",
953
+ },
954
+ {
955
+ "id": "promote-redteam-finding",
956
+ "command": "promote_to_regression",
957
+ "path": _suite_path_text(finding_path),
958
+ "name": f"{suite_name}-promoted-regression",
959
+ "min_level": str(min_level),
960
+ "max_findings": int(max_findings),
961
+ },
962
+ {
963
+ "id": "replay-regression-manifest",
964
+ "command": "replay",
965
+ "path": replay_paths[0],
966
+ "manifests": replay_paths,
967
+ "name": f"{suite_name}-replay",
968
+ },
969
+ ]
970
+ return build_suite_manifest(
971
+ name=suite_name,
972
+ required_env=required_env,
973
+ jobs=jobs,
974
+ required_capabilities={
975
+ "commands": [
976
+ "baseline",
977
+ "compare",
978
+ "report",
979
+ "promote_to_regression",
980
+ "replay",
981
+ ],
982
+ "result_kinds": [
983
+ "agent_learning.baseline.v1",
984
+ "agent_learning.compare.v1",
985
+ "agent_learning.report.v1",
986
+ "agent_learning.regression_promotion.v1",
987
+ "agent_learning.replay.v1",
988
+ ],
989
+ "metrics": [
990
+ "compare_score_delta",
991
+ "replay_pass_rate",
992
+ ],
993
+ },
994
+ metadata={
995
+ "source": "fi.alk.suite.build_regression_artifact_suite_manifest",
996
+ "task_kind": "regression_artifact_lifecycle",
997
+ **copy.deepcopy(dict(metadata or {})),
998
+ },
999
+ )
1000
+
1001
+
1002
+ def build_optimization_lifecycle_plan(
1003
+ *,
1004
+ optimize_manifest_path: str | Path,
1005
+ workspace_dir: str | Path | None = None,
1006
+ name: str = "optimization-lifecycle",
1007
+ required_env: Sequence[str] = (),
1008
+ frozen_profile_path: str | Path | None = None,
1009
+ ) -> dict[str, Any]:
1010
+ """Build an executable optimize -> promote -> replay lifecycle plan.
1011
+
1012
+ When ``frozen_profile_path`` names a frozen capability-profile contract
1013
+ (kind ``agent-learning.frozen-capability-profile.v1``, ARCH §2a), the plan
1014
+ gains a ``replay_frozen_profile`` step between the promotion and the
1015
+ regression replay: every frozen row is re-closed against the optimization
1016
+ artifact and an improving-but-row-breaking candidate is vetoed
1017
+ (hetvabhasa class ``badhita``) before any replay runs.
1018
+ """
1019
+
1020
+ paths = _optimization_lifecycle_paths(
1021
+ optimize_manifest_path=optimize_manifest_path,
1022
+ workspace_dir=workspace_dir,
1023
+ )
1024
+ if frozen_profile_path is not None:
1025
+ frozen_path = Path(frozen_profile_path).expanduser().resolve()
1026
+ paths["frozen_profile"] = frozen_path
1027
+ paths["frozen_profile_replay"] = (
1028
+ paths["optimization"].parent / "frozen-profile-replay.json"
1029
+ )
1030
+ required_env_args = _required_env_cli_args(required_env)
1031
+ steps = [
1032
+ _lifecycle_step(
1033
+ "dry_run_optimization",
1034
+ "Dry Run Optimization",
1035
+ ["agent-learn", "optimize", paths["optimize_manifest"], "--dry-run"],
1036
+ ),
1037
+ _lifecycle_step(
1038
+ "optimize",
1039
+ "Run Optimization",
1040
+ [
1041
+ "agent-learn",
1042
+ "optimize",
1043
+ paths["optimize_manifest"],
1044
+ "--output",
1045
+ paths["optimization"],
1046
+ "--junit",
1047
+ paths["optimization_junit"],
1048
+ "--sarif",
1049
+ paths["optimization_sarif"],
1050
+ "--markdown",
1051
+ paths["optimization_markdown"],
1052
+ ],
1053
+ outputs={
1054
+ "json": paths["optimization"],
1055
+ "junit": paths["optimization_junit"],
1056
+ "sarif": paths["optimization_sarif"],
1057
+ "markdown": paths["optimization_markdown"],
1058
+ },
1059
+ ),
1060
+ _lifecycle_step(
1061
+ "report_optimization",
1062
+ "Report Optimization",
1063
+ [
1064
+ "agent-learn",
1065
+ "report",
1066
+ paths["optimization"],
1067
+ "--output",
1068
+ paths["optimization_report"],
1069
+ "--markdown",
1070
+ paths["optimization_report_markdown"],
1071
+ ],
1072
+ outputs={
1073
+ "json": paths["optimization_report"],
1074
+ "markdown": paths["optimization_report_markdown"],
1075
+ },
1076
+ ),
1077
+ _lifecycle_step(
1078
+ "promote_to_regression",
1079
+ "Promote To Regression",
1080
+ [
1081
+ "agent-learn",
1082
+ "promote-to-regression",
1083
+ paths["optimization"],
1084
+ "--output",
1085
+ paths["promotion"],
1086
+ "--manifest",
1087
+ paths["regression_manifest"],
1088
+ "--min-level",
1089
+ "note",
1090
+ "--max-findings",
1091
+ "1",
1092
+ *required_env_args,
1093
+ ],
1094
+ outputs={
1095
+ "json": paths["promotion"],
1096
+ "manifest": paths["regression_manifest"],
1097
+ },
1098
+ ),
1099
+ _lifecycle_step(
1100
+ "report_promotion",
1101
+ "Report Promotion",
1102
+ [
1103
+ "agent-learn",
1104
+ "report",
1105
+ paths["promotion"],
1106
+ "--output",
1107
+ paths["promotion_report"],
1108
+ "--markdown",
1109
+ paths["promotion_report_markdown"],
1110
+ ],
1111
+ outputs={
1112
+ "json": paths["promotion_report"],
1113
+ "markdown": paths["promotion_report_markdown"],
1114
+ },
1115
+ ),
1116
+ *(
1117
+ [
1118
+ _lifecycle_step(
1119
+ "replay_frozen_profile",
1120
+ "Replay Frozen Capability Profile",
1121
+ [
1122
+ sys.executable,
1123
+ "-c",
1124
+ (
1125
+ "import json, pathlib; "
1126
+ "from fi.alk import optimize; "
1127
+ "result = json.loads(pathlib.Path("
1128
+ f"{str(paths['optimization'])!r}"
1129
+ ").read_text(encoding='utf-8')); "
1130
+ "frozen = json.loads(pathlib.Path("
1131
+ f"{str(paths['frozen_profile'])!r}"
1132
+ ").read_text(encoding='utf-8')); "
1133
+ "verdict = optimize.replay_frozen_profile(result, frozen); "
1134
+ "pathlib.Path("
1135
+ f"{str(paths['frozen_profile_replay'])!r}"
1136
+ ").write_text(json.dumps(verdict, indent=2, "
1137
+ "sort_keys=True, default=str), encoding='utf-8'); "
1138
+ "raise SystemExit(1 if verdict.get('veto') else 0)"
1139
+ ),
1140
+ ],
1141
+ outputs={"json": paths["frozen_profile_replay"]},
1142
+ )
1143
+ ]
1144
+ if frozen_profile_path is not None
1145
+ else []
1146
+ ),
1147
+ _lifecycle_step(
1148
+ "replay_regression",
1149
+ "Replay Regression",
1150
+ [
1151
+ "agent-learn",
1152
+ "replay",
1153
+ paths["regression_manifest"],
1154
+ "--output",
1155
+ paths["replay"],
1156
+ "--junit",
1157
+ paths["replay_junit"],
1158
+ "--sarif",
1159
+ paths["replay_sarif"],
1160
+ "--markdown",
1161
+ paths["replay_markdown"],
1162
+ ],
1163
+ outputs={
1164
+ "json": paths["replay"],
1165
+ "junit": paths["replay_junit"],
1166
+ "sarif": paths["replay_sarif"],
1167
+ "markdown": paths["replay_markdown"],
1168
+ },
1169
+ ),
1170
+ _lifecycle_step(
1171
+ "report_replay",
1172
+ "Report Replay",
1173
+ [
1174
+ "agent-learn",
1175
+ "report",
1176
+ paths["replay"],
1177
+ "--output",
1178
+ paths["replay_report"],
1179
+ "--markdown",
1180
+ paths["replay_report_markdown"],
1181
+ ],
1182
+ outputs={
1183
+ "json": paths["replay_report"],
1184
+ "markdown": paths["replay_report_markdown"],
1185
+ },
1186
+ ),
1187
+ ]
1188
+ return {
1189
+ "kind": AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND,
1190
+ "name": str(name),
1191
+ "required_env": _unique_strings(required_env),
1192
+ "artifacts": {key: str(value) for key, value in paths.items()},
1193
+ "steps": steps,
1194
+ "metadata": {
1195
+ "source": "fi.alk.suite.build_optimization_lifecycle_plan",
1196
+ "research_synthesis": (
1197
+ "Deterministic optimization transactions: diagnose/search, "
1198
+ "export, promote, replay, and expose action cards over one "
1199
+ "shared evidence trail."
1200
+ ),
1201
+ },
1202
+ }
1203
+
1204
+
1205
+ def run_optimization_lifecycle_file(
1206
+ optimize_manifest_path: str | Path,
1207
+ *,
1208
+ workspace_dir: str | Path | None = None,
1209
+ name: str = "optimization-lifecycle",
1210
+ required_env: Sequence[str] = (),
1211
+ ) -> dict[str, Any]:
1212
+ """Run optimize, report, promote, replay, and report replay via SDK."""
1213
+
1214
+ from fi.alk import optimize, simulate
1215
+
1216
+ plan = build_optimization_lifecycle_plan(
1217
+ optimize_manifest_path=optimize_manifest_path,
1218
+ workspace_dir=workspace_dir,
1219
+ name=name,
1220
+ required_env=required_env,
1221
+ )
1222
+ paths = {key: Path(value) for key, value in plan["artifacts"].items()}
1223
+ outputs_written: list[str] = []
1224
+
1225
+ optimization = optimize.optimize_manifest_file(paths["optimize_manifest"])
1226
+ outputs_written.extend(
1227
+ _write_lifecycle_result_bundle(
1228
+ optimization,
1229
+ json_path=paths["optimization"],
1230
+ junit_path=paths["optimization_junit"],
1231
+ sarif_path=paths["optimization_sarif"],
1232
+ markdown_path=paths["optimization_markdown"],
1233
+ source_path=paths["optimize_manifest"],
1234
+ )
1235
+ )
1236
+
1237
+ optimization_report = simulate.render_report(
1238
+ optimization,
1239
+ source_path=paths["optimization"],
1240
+ )
1241
+ outputs_written.extend(
1242
+ _write_lifecycle_report_bundle(
1243
+ optimization_report,
1244
+ json_path=paths["optimization_report"],
1245
+ markdown_path=paths["optimization_report_markdown"],
1246
+ source_path=paths["optimization"],
1247
+ )
1248
+ )
1249
+
1250
+ promotion = simulate.promote_to_regression(
1251
+ optimization,
1252
+ source_path=paths["optimization"],
1253
+ min_level="note",
1254
+ max_findings=1,
1255
+ required_env=required_env,
1256
+ )
1257
+ outputs_written.append(_write_json(paths["promotion"], promotion))
1258
+ manifest = promotion.get("manifest")
1259
+ if isinstance(manifest, Mapping):
1260
+ outputs_written.append(_write_json(paths["regression_manifest"], manifest))
1261
+
1262
+ promotion_report = simulate.render_report(
1263
+ promotion,
1264
+ source_path=paths["promotion"],
1265
+ )
1266
+ outputs_written.extend(
1267
+ _write_lifecycle_report_bundle(
1268
+ promotion_report,
1269
+ json_path=paths["promotion_report"],
1270
+ markdown_path=paths["promotion_report_markdown"],
1271
+ source_path=paths["promotion"],
1272
+ )
1273
+ )
1274
+
1275
+ replay = simulate.replay_manifests([paths["regression_manifest"]])
1276
+ outputs_written.extend(
1277
+ _write_lifecycle_result_bundle(
1278
+ replay,
1279
+ json_path=paths["replay"],
1280
+ junit_path=paths["replay_junit"],
1281
+ sarif_path=paths["replay_sarif"],
1282
+ markdown_path=paths["replay_markdown"],
1283
+ source_path=paths["regression_manifest"],
1284
+ )
1285
+ )
1286
+
1287
+ replay_report = simulate.render_report(replay, source_path=paths["replay"])
1288
+ outputs_written.extend(
1289
+ _write_lifecycle_report_bundle(
1290
+ replay_report,
1291
+ json_path=paths["replay_report"],
1292
+ markdown_path=paths["replay_report_markdown"],
1293
+ source_path=paths["replay"],
1294
+ )
1295
+ )
1296
+
1297
+ passed = all(
1298
+ payload.get("status") == "passed"
1299
+ for payload in (optimization, promotion, replay)
1300
+ )
1301
+ return {
1302
+ "kind": AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND,
1303
+ "name": str(name),
1304
+ "status": "passed" if passed else "failed",
1305
+ "exit_code": 0 if passed else 1,
1306
+ "summary": {
1307
+ "optimization_score": dict(optimization.get("summary") or {}).get(
1308
+ "optimization_score"
1309
+ ),
1310
+ "promotion_kind": dict(promotion.get("summary") or {}).get(
1311
+ "promotion_kind"
1312
+ ),
1313
+ "promoted_manifest_count": dict(promotion.get("summary") or {}).get(
1314
+ "promoted_manifest_count"
1315
+ ),
1316
+ "replay_pass_rate": dict(replay.get("summary") or {}).get(
1317
+ "replay_pass_rate"
1318
+ ),
1319
+ "step_count": len(plan["steps"]),
1320
+ "outputs_written_count": len(outputs_written),
1321
+ },
1322
+ "plan": plan,
1323
+ "artifacts": {
1324
+ "optimization": optimization,
1325
+ "optimization_report": optimization_report,
1326
+ "promotion": promotion,
1327
+ "promotion_report": promotion_report,
1328
+ "replay": replay,
1329
+ "replay_report": replay_report,
1330
+ },
1331
+ "outputs_written": outputs_written,
1332
+ }
1333
+
1334
+
1335
+ def write_suite_file(manifest: Mapping[str, Any], path: str | Path) -> Path:
1336
+ """Write a suite manifest as formatted JSON and return the resolved path."""
1337
+
1338
+ suite_path = Path(path).expanduser().resolve()
1339
+ suite_path.parent.mkdir(parents=True, exist_ok=True)
1340
+ suite_path.write_text(
1341
+ json.dumps(dict(manifest), indent=2, sort_keys=True, default=str) + "\n",
1342
+ encoding="utf-8",
1343
+ )
1344
+ return suite_path
1345
+
1346
+
1347
+ def _write_suite_json(payload: Mapping[str, Any], path: str | Path) -> Path:
1348
+ output_path = Path(path).expanduser().resolve()
1349
+ output_path.parent.mkdir(parents=True, exist_ok=True)
1350
+ output_path.write_text(
1351
+ json.dumps(dict(payload), indent=2, sort_keys=True, default=str) + "\n",
1352
+ encoding="utf-8",
1353
+ )
1354
+ return output_path
1355
+
1356
+
1357
+ def run_suite_file(
1358
+ path: str | Path,
1359
+ *,
1360
+ options: Optional[SuiteRunOptions] = None,
1361
+ name: Optional[str] = None,
1362
+ threshold: Optional[float] = None,
1363
+ max_candidates: Optional[int] = None,
1364
+ dry_run: Optional[bool] = None,
1365
+ fail_fast: Optional[bool] = None,
1366
+ require_optimizer_governance: Optional[bool] = None,
1367
+ ) -> dict[str, Any]:
1368
+ suite_path = Path(path).expanduser().resolve()
1369
+ suite = load_suite_file(suite_path)
1370
+ return run_suite(
1371
+ suite,
1372
+ suite_path=suite_path,
1373
+ options=_merge_options(
1374
+ options,
1375
+ name=name,
1376
+ threshold=threshold,
1377
+ max_candidates=max_candidates,
1378
+ dry_run=dry_run,
1379
+ fail_fast=fail_fast,
1380
+ require_optimizer_governance=require_optimizer_governance,
1381
+ ),
1382
+ )
1383
+
1384
+
1385
+ def run_suite(
1386
+ suite: Mapping[str, Any],
1387
+ *,
1388
+ suite_path: str | Path = ".",
1389
+ options: Optional[SuiteRunOptions] = None,
1390
+ name: Optional[str] = None,
1391
+ threshold: Optional[float] = None,
1392
+ max_candidates: Optional[int] = None,
1393
+ dry_run: Optional[bool] = None,
1394
+ fail_fast: Optional[bool] = None,
1395
+ require_optimizer_governance: Optional[bool] = None,
1396
+ ) -> dict[str, Any]:
1397
+ started = time.time()
1398
+ opts = _merge_options(
1399
+ options,
1400
+ name=name,
1401
+ threshold=threshold,
1402
+ max_candidates=max_candidates,
1403
+ dry_run=dry_run,
1404
+ fail_fast=fail_fast,
1405
+ require_optimizer_governance=require_optimizer_governance,
1406
+ )
1407
+ suite_path = Path(suite_path).expanduser().resolve()
1408
+ base_dir = _suite_base_dir(suite_path)
1409
+ runtime_suite = _prepare_suite(copy.deepcopy(dict(suite)), base_dir=base_dir)
1410
+ if opts.require_optimizer_governance:
1411
+ optimizer_policy = _suite_optimizer_governance_policy(runtime_suite)
1412
+ optimizer_policy["require_optimizer_governance"] = True
1413
+ optimizer_policy["require_passed"] = True
1414
+ optimizer_policy["min_governed"] = max(
1415
+ int(optimizer_policy.get("min_governed") or 0),
1416
+ 1,
1417
+ )
1418
+ runtime_suite["optimizer_governance_policy"] = {
1419
+ **optimizer_policy,
1420
+ }
1421
+ validate_suite_env(runtime_suite, suite_path=suite_path)
1422
+
1423
+ children: list[dict[str, Any]] = []
1424
+ for index, job in enumerate(_suite_jobs(runtime_suite), start=1):
1425
+ child = _execute_job(
1426
+ job,
1427
+ index=index,
1428
+ base_dir=base_dir,
1429
+ suite_options=opts,
1430
+ )
1431
+ children.append(child)
1432
+ if int(child.get("exit_code", 1)) != 0 and opts.fail_fast:
1433
+ break
1434
+
1435
+ payload = _suite_result(
1436
+ suite=runtime_suite,
1437
+ suite_path=suite_path,
1438
+ children=children,
1439
+ name=opts.name,
1440
+ dry_run=opts.dry_run,
1441
+ fail_fast=opts.fail_fast,
1442
+ duration_seconds=round(time.time() - started, 4),
1443
+ )
1444
+ return public_payload(payload, kind=AGENT_LEARNING_SUITE_KIND)
1445
+
1446
+
1447
+ def optimize_suite_file(
1448
+ path: str | Path,
1449
+ *,
1450
+ options: Optional[SuiteOptimizationOptions] = None,
1451
+ name: Optional[str] = None,
1452
+ threshold: Optional[float] = None,
1453
+ max_candidates: Optional[int] = None,
1454
+ dry_run: Optional[bool] = None,
1455
+ ) -> dict[str, Any]:
1456
+ """Load and optimize a full Agent Learning suite."""
1457
+
1458
+ suite_path = Path(path).expanduser().resolve()
1459
+ suite = load_suite_file(suite_path)
1460
+ return optimize_suite(
1461
+ suite,
1462
+ suite_path=suite_path,
1463
+ options=_merge_optimization_options(
1464
+ options,
1465
+ name=name,
1466
+ threshold=threshold,
1467
+ max_candidates=max_candidates,
1468
+ dry_run=dry_run,
1469
+ ),
1470
+ )
1471
+
1472
+
1473
+ def optimize_suite(
1474
+ suite: Mapping[str, Any],
1475
+ *,
1476
+ suite_path: str | Path = ".",
1477
+ options: Optional[SuiteOptimizationOptions] = None,
1478
+ name: Optional[str] = None,
1479
+ threshold: Optional[float] = None,
1480
+ max_candidates: Optional[int] = None,
1481
+ dry_run: Optional[bool] = None,
1482
+ ) -> dict[str, Any]:
1483
+ """Optimize a mixed Agent Learning suite and return a unified artifact."""
1484
+
1485
+ started = time.time()
1486
+ opts = _merge_optimization_options(
1487
+ options,
1488
+ name=name,
1489
+ threshold=threshold,
1490
+ max_candidates=max_candidates,
1491
+ dry_run=dry_run,
1492
+ )
1493
+ suite_path = Path(suite_path).expanduser().resolve()
1494
+ base_dir = _suite_base_dir(suite_path)
1495
+ runtime_suite = copy.deepcopy(dict(suite))
1496
+ if opts.name:
1497
+ runtime_suite["name"] = opts.name
1498
+ if opts.threshold is not None:
1499
+ runtime_suite.setdefault("optimization", {})["threshold"] = opts.threshold
1500
+ if opts.max_candidates is not None:
1501
+ runtime_suite.setdefault("optimization", {}).setdefault(
1502
+ "optimizer", {}
1503
+ )["max_candidates"] = opts.max_candidates
1504
+
1505
+ prepared = _prepare_suite(runtime_suite, base_dir=base_dir)
1506
+ validate_suite_env(prepared, suite_path=suite_path)
1507
+ cli = _optimization_cli()
1508
+ optimization = cli._optimization_config(prepared)
1509
+ target_config = cli._target_config(optimization)
1510
+ optimizer_config = cli._optimizer_config(optimization)
1511
+ if opts.dry_run:
1512
+ return public_payload({
1513
+ "schema_version": AGENT_LEARNING_CLI_SCHEMA_VERSION,
1514
+ "kind": AGENT_LEARNING_SUITE_OPTIMIZATION_KIND,
1515
+ "name": str(prepared.get("name") or suite_path.stem),
1516
+ "status": "passed",
1517
+ "exit_code": 0,
1518
+ "dry_run": True,
1519
+ "summary": {
1520
+ "job_count": len(_suite_jobs(prepared)),
1521
+ "required_env": required_suite_env(prepared, suite_path=suite_path),
1522
+ "search_path_count": len(target_config.get("search_space", {})),
1523
+ "max_candidates": optimizer_config.get("max_candidates"),
1524
+ },
1525
+ "duration_seconds": round(time.time() - started, 4),
1526
+ }, kind=AGENT_LEARNING_SUITE_OPTIMIZATION_KIND)
1527
+
1528
+ try:
1529
+ from fi.alk import optimize as agent_optimize
1530
+ except Exception as exc: # pragma: no cover - optional dependency clarity
1531
+ raise SuiteError(
1532
+ "Agent Learning Kit optimizer engine is required for suite optimization."
1533
+ ) from exc
1534
+
1535
+ problem = agent_optimize.problem_from_agent_learning_suite(
1536
+ prepared,
1537
+ suite_path=suite_path,
1538
+ name=str(prepared.get("name") or suite_path.stem),
1539
+ )
1540
+ optimization_result = problem.optimize()
1541
+ payload = cli._optimization_result(
1542
+ manifest=prepared,
1543
+ manifest_path=suite_path,
1544
+ optimization_result=optimization_result,
1545
+ threshold=float(optimization.get("threshold", 1.0)),
1546
+ duration_seconds=round(time.time() - started, 4),
1547
+ )
1548
+ payload["kind"] = AGENT_LEARNING_SUITE_OPTIMIZATION_KIND
1549
+ payload["suite"] = _suite_descriptor(prepared)
1550
+ payload["optimization"]["source"] = "agent_learning_suite"
1551
+ if "manifest_optimization" in payload["optimization"]:
1552
+ artifact = copy.deepcopy(payload["optimization"]["manifest_optimization"])
1553
+ artifact["kind"] = "agent_learning_suite_optimization"
1554
+ artifact["source"] = "agent_learning_suite"
1555
+ payload["optimization"]["suite_optimization"] = artifact
1556
+ payload["summary"]["job_count"] = len(_suite_jobs(prepared))
1557
+ payload["summary"]["child_command_count"] = _suite_job_command_counts(prepared)
1558
+ action_plan = _artifact_action_plan_card(payload)
1559
+ if action_plan is not None:
1560
+ payload["artifact_action_plan"] = action_plan
1561
+ payload["optimization"]["artifact_action_plan"] = copy.deepcopy(action_plan)
1562
+ payload["summary"]["artifact_action_best_action_id"] = action_plan.get(
1563
+ "selected_action_id"
1564
+ )
1565
+ return public_payload(payload, kind=AGENT_LEARNING_SUITE_OPTIMIZATION_KIND)
1566
+
1567
+
1568
+ def render_junit(result: Mapping[str, Any]) -> str:
1569
+ name = escape(str(result.get("name") or "agent-learning-suite"))
1570
+ children = list(result.get("children") or result.get("jobs") or [])
1571
+ finding_failures = [
1572
+ finding
1573
+ for finding in list(result.get("findings") or [])
1574
+ if str(_as_mapping(finding).get("type"))
1575
+ in {
1576
+ "suite_required_capability_missing",
1577
+ "suite_evidence_admission_missing",
1578
+ "suite_evidence_freeze_missing",
1579
+ "suite_framework_adapter_conformance_failed",
1580
+ "suite_framework_coverage_missing",
1581
+ "suite_optimizer_governance_failed",
1582
+ "suite_optimizer_governance_missing",
1583
+ "suite_optimizer_governance_warning",
1584
+ }
1585
+ ]
1586
+ failures = (
1587
+ sum(1 for child in children if int(child.get("exit_code", 1)) != 0)
1588
+ + len(finding_failures)
1589
+ )
1590
+ lines = [
1591
+ (
1592
+ f'<testsuite name="{name}" tests="{len(children) + len(finding_failures)}" '
1593
+ f'failures="{failures}" errors="0">'
1594
+ )
1595
+ ]
1596
+ for child in children:
1597
+ child_name = escape(str(child.get("id") or child.get("name") or "job"))
1598
+ class_name = escape(str(child.get("command") or "suite"))
1599
+ duration = float(child.get("duration_seconds") or 0.0)
1600
+ lines.append(
1601
+ f' <testcase classname="{class_name}" name="{child_name}" '
1602
+ f'time="{duration:.4f}">'
1603
+ )
1604
+ if int(child.get("exit_code", 1)) != 0:
1605
+ message = escape(str(child.get("error") or child.get("status") or "failed"))
1606
+ lines.append(f' <failure message="{message}">{message}</failure>')
1607
+ lines.append(" </testcase>")
1608
+ for index, finding in enumerate(finding_failures, start=1):
1609
+ item = _as_mapping(finding)
1610
+ finding_name = escape(str(item.get("type") or f"suite_finding_{index}"))
1611
+ message = escape(str(item.get("reason") or finding_name))
1612
+ lines.append(f' <testcase classname="suite" name="{finding_name}" time="0.0000">')
1613
+ lines.append(f' <failure message="{message}">{message}</failure>')
1614
+ lines.append(" </testcase>")
1615
+ lines.append("</testsuite>")
1616
+ return "\n".join(lines)
1617
+
1618
+
1619
+ def render_sarif(
1620
+ result: Mapping[str, Any],
1621
+ *,
1622
+ manifest_path: str | Path = ".",
1623
+ ) -> str:
1624
+ suite_path = Path(manifest_path).expanduser().resolve()
1625
+ findings = _suite_sarif_findings(result)
1626
+ sarif_results = []
1627
+ for finding in findings:
1628
+ rule_id = str(finding.get("type") or finding.get("rule_id") or "suite_finding")
1629
+ level = str(finding.get("level") or finding.get("severity") or "error").lower()
1630
+ if level not in {"none", "note", "warning", "error"}:
1631
+ level = "warning"
1632
+ location_path = str(finding.get("path") or suite_path)
1633
+ sarif_results.append(
1634
+ {
1635
+ "ruleId": rule_id,
1636
+ "level": level,
1637
+ "message": {"text": str(finding.get("reason") or rule_id)},
1638
+ "locations": [
1639
+ {
1640
+ "physicalLocation": {
1641
+ "artifactLocation": {"uri": location_path},
1642
+ }
1643
+ }
1644
+ ],
1645
+ }
1646
+ )
1647
+ payload = {
1648
+ "$schema": "https://json.schemastore.org/sarif-2.1.0.json",
1649
+ "version": "2.1.0",
1650
+ "runs": [
1651
+ {
1652
+ "tool": {
1653
+ "driver": {
1654
+ "name": "agent-learning-suite",
1655
+ "informationUri": "https://futureagi.com",
1656
+ "rules": [],
1657
+ }
1658
+ },
1659
+ "results": sarif_results,
1660
+ }
1661
+ ],
1662
+ }
1663
+ return json.dumps(payload, indent=2, sort_keys=True)
1664
+
1665
+
1666
+ def render_markdown(
1667
+ result: Mapping[str, Any],
1668
+ *,
1669
+ source_path: str | Path = ".",
1670
+ ) -> str:
1671
+ summary = dict(result.get("summary") or {})
1672
+ certificate = _as_mapping(result.get("trust_certificate"))
1673
+ lines = [
1674
+ f"# {result.get('name') or 'agent-learning-suite'}",
1675
+ "",
1676
+ f"- Source: `{Path(source_path)}`",
1677
+ f"- Status: `{result.get('status')}`",
1678
+ f"- Jobs: {summary.get('passed_count', 0)}/{summary.get('job_count', 0)} passed",
1679
+ f"- Score: {summary.get('score', 0.0)}",
1680
+ (
1681
+ "- Trust Certificate: "
1682
+ f"{certificate.get('verdict') or summary.get('trust_certificate_verdict')}"
1683
+ f" ({certificate.get('assurance_level') or summary.get('trust_certificate_assurance_level')})"
1684
+ ),
1685
+ (
1686
+ "- Evidence: "
1687
+ f"{summary.get('admitted_evidence_count', 0)} admitted, "
1688
+ f"{summary.get('non_admitted_evidence_count', 0)} non-admitted, "
1689
+ f"{summary.get('rejected_evidence_count', 0)} rejected, "
1690
+ f"{summary.get('frozen_evidence_count', 0)} frozen"
1691
+ ),
1692
+ (
1693
+ "- Frameworks: "
1694
+ f"{summary.get('observed_framework_count', 0)} observed, "
1695
+ f"{summary.get('missing_framework_count', 0)} missing, "
1696
+ f"{summary.get('adapter_conformance_failed_count', 0)} adapter-failed"
1697
+ ),
1698
+ "",
1699
+ "## Trust Certificate",
1700
+ "",
1701
+ f"- Verdict: `{certificate.get('verdict')}`",
1702
+ f"- Assurance Level: `{certificate.get('assurance_level')}`",
1703
+ f"- Promotion Ready: `{certificate.get('promotion_ready')}`",
1704
+ f"- Reason: {certificate.get('reason') or ''}",
1705
+ "",
1706
+ "| Gate | Status | Required |",
1707
+ "| --- | --- | --- |",
1708
+ ]
1709
+ for gate in _as_list(certificate.get("gates")):
1710
+ gate_item = _as_mapping(gate)
1711
+ if not gate_item:
1712
+ continue
1713
+ lines.append(
1714
+ "| "
1715
+ f"{_md_cell(gate_item.get('id') or '')} | "
1716
+ f"{_md_cell(gate_item.get('status') or '')} | "
1717
+ f"{_md_cell(str(bool(gate_item.get('required'))))} |"
1718
+ )
1719
+ lines.extend([
1720
+ "",
1721
+ "| Job | Command | Status | Evidence | Exit |",
1722
+ "| --- | --- | --- | --- | --- |",
1723
+ ])
1724
+ for child in list(result.get("children") or result.get("jobs") or []):
1725
+ evidence = _as_mapping(child.get("evidence"))
1726
+ evidence_cell = evidence.get("status") or ""
1727
+ if evidence.get("role") and evidence.get("role") != evidence_cell:
1728
+ evidence_cell = f"{evidence_cell} ({evidence.get('role')})"
1729
+ lines.append(
1730
+ "| "
1731
+ f"{_md_cell(child.get('id') or child.get('name') or '')} | "
1732
+ f"{_md_cell(child.get('command') or '')} | "
1733
+ f"{_md_cell(child.get('status') or '')} | "
1734
+ f"{_md_cell(evidence_cell)} | "
1735
+ f"{int(child.get('exit_code', 1))} |"
1736
+ )
1737
+ return "\n".join(lines) + "\n"
1738
+
1739
+
1740
+ def _prepare_suite(suite: dict[str, Any], *, base_dir: Path) -> dict[str, Any]:
1741
+ jobs = _as_list(suite.get("jobs") or suite.get("runs") or suite.get("steps"))
1742
+ if not jobs:
1743
+ raise SuiteError("suite manifest requires at least one job")
1744
+ prepared_jobs = []
1745
+ for index, job in enumerate(jobs, start=1):
1746
+ if not isinstance(job, Mapping):
1747
+ raise SuiteError(f"suite job[{index}] must be an object")
1748
+ prepared = dict(job)
1749
+ prepared["command"] = _normalize_command(
1750
+ prepared.get("command") or prepared.get("type") or prepared.get("kind")
1751
+ )
1752
+ prepared.setdefault("id", f"{prepared['command']}-{index}")
1753
+ _job_path(prepared, base_dir=base_dir)
1754
+ prepared_jobs.append(prepared)
1755
+ suite["jobs"] = prepared_jobs
1756
+ suite.setdefault("version", AGENT_LEARNING_SUITE_KIND)
1757
+ suite.setdefault("name", "agent-learning-suite")
1758
+ return suite
1759
+
1760
+
1761
+ def _execute_job(
1762
+ job: Mapping[str, Any],
1763
+ *,
1764
+ index: int,
1765
+ base_dir: Path,
1766
+ suite_options: SuiteRunOptions,
1767
+ ) -> dict[str, Any]:
1768
+ started = time.time()
1769
+ command = _normalize_command(job.get("command") or job.get("type"))
1770
+ path = _job_path(job, base_dir=base_dir)
1771
+ job_id = str(job.get("id") or f"{command}-{index}")
1772
+ try:
1773
+ payload = _execute_child_payload(
1774
+ command,
1775
+ path=path,
1776
+ base_dir=base_dir,
1777
+ job=job,
1778
+ suite_options=suite_options,
1779
+ )
1780
+ payload = copy.deepcopy(dict(payload))
1781
+ outputs_written = _write_child_outputs(
1782
+ payload,
1783
+ command=command,
1784
+ job=job,
1785
+ path=path,
1786
+ )
1787
+ payload["outputs_written"] = outputs_written
1788
+ result = {
1789
+ "id": job_id,
1790
+ "command": command,
1791
+ "path": str(path),
1792
+ "kind": payload.get("kind"),
1793
+ "name": payload.get("name"),
1794
+ "status": str(payload.get("status") or "unknown"),
1795
+ "exit_code": int(payload.get("exit_code", 1)),
1796
+ "summary": copy.deepcopy(dict(payload.get("summary") or {})),
1797
+ "findings": copy.deepcopy(list(payload.get("findings") or [])),
1798
+ "outputs_written": outputs_written,
1799
+ "duration_seconds": round(time.time() - started, 4),
1800
+ "result": payload,
1801
+ }
1802
+ result["evidence"] = _suite_child_evidence(
1803
+ job,
1804
+ result,
1805
+ base_dir=base_dir,
1806
+ )
1807
+ return result
1808
+ except Exception as exc:
1809
+ result = {
1810
+ "id": job_id,
1811
+ "command": command,
1812
+ "path": str(path),
1813
+ "kind": None,
1814
+ "name": job.get("name"),
1815
+ "status": "failed",
1816
+ "exit_code": 1,
1817
+ "summary": {},
1818
+ "findings": [
1819
+ {
1820
+ "type": "suite_child_failed",
1821
+ "level": "error",
1822
+ "reason": str(exc),
1823
+ "job": job_id,
1824
+ "command": command,
1825
+ "path": str(path),
1826
+ }
1827
+ ],
1828
+ "outputs_written": [],
1829
+ "duration_seconds": round(time.time() - started, 4),
1830
+ "error": str(exc),
1831
+ }
1832
+ result["evidence"] = _suite_child_evidence(
1833
+ job,
1834
+ result,
1835
+ base_dir=base_dir,
1836
+ )
1837
+ return result
1838
+
1839
+
1840
+ def _execute_child_payload(
1841
+ command: str,
1842
+ *,
1843
+ path: Path,
1844
+ base_dir: Path,
1845
+ job: Mapping[str, Any],
1846
+ suite_options: SuiteRunOptions,
1847
+ ) -> dict[str, Any]:
1848
+ if command == "run":
1849
+ from fi.alk import simulate
1850
+ from fi.alk.cli import AGENT_LEARNING_RUN_KIND
1851
+
1852
+ payload = _run_async(
1853
+ simulate.run_manifest_file(
1854
+ path,
1855
+ name=_job_name(job),
1856
+ threshold=_job_threshold(job, suite_options),
1857
+ no_eval=bool(job.get("no_eval", job.get("no-eval", False))),
1858
+ dry_run=_job_dry_run(job, suite_options),
1859
+ )
1860
+ )
1861
+ payload["kind"] = AGENT_LEARNING_RUN_KIND
1862
+ return payload
1863
+ if command == "suite":
1864
+ payload = run_suite_file(
1865
+ path,
1866
+ options=SuiteRunOptions(
1867
+ name=_job_name(job),
1868
+ threshold=_job_threshold(job, suite_options),
1869
+ max_candidates=_job_max_candidates(job, suite_options),
1870
+ dry_run=_job_dry_run(job, suite_options),
1871
+ fail_fast=bool(
1872
+ suite_options.fail_fast
1873
+ or job.get("fail_fast")
1874
+ or job.get("fail-fast")
1875
+ ),
1876
+ require_optimizer_governance=suite_options.require_optimizer_governance,
1877
+ ),
1878
+ )
1879
+ payload["kind"] = AGENT_LEARNING_SUITE_KIND
1880
+ return payload
1881
+ if command == "action_run":
1882
+ from fi.alk import actions
1883
+
1884
+ artifact = actions.load_artifact_file(path)
1885
+ return actions.run_action(
1886
+ artifact,
1887
+ _job_action_id(job),
1888
+ source_path=path,
1889
+ inputs=_job_action_inputs(job),
1890
+ cwd=_job_action_cwd(job, base_dir=base_dir),
1891
+ dry_run=_job_dry_run(job, suite_options),
1892
+ name=_job_name(job),
1893
+ artifact_output_path=_job_action_artifact_output(job),
1894
+ )
1895
+ if command == "eval":
1896
+ from fi.alk import evals
1897
+ from fi.alk.cli import AGENT_LEARNING_EVAL_KIND
1898
+
1899
+ payload = evals.run_eval_suite_file(
1900
+ path,
1901
+ name=_job_name(job),
1902
+ threshold=_job_threshold(job, suite_options),
1903
+ dry_run=_job_dry_run(job, suite_options),
1904
+ )
1905
+ payload["kind"] = AGENT_LEARNING_EVAL_KIND
1906
+ return payload
1907
+ if command == "eval_artifact":
1908
+ from fi.alk import evals
1909
+ from fi.alk.cli import AGENT_LEARNING_ARTIFACT_EVAL_KIND
1910
+
1911
+ config_path = _job_optional_path(
1912
+ job,
1913
+ base_dir=base_dir,
1914
+ keys=("config", "eval_config", "agent_report_config"),
1915
+ )
1916
+ config = evals.load_artifact_file(config_path) if config_path else None
1917
+ payload = evals.evaluate_artifact_file(
1918
+ path,
1919
+ config=config,
1920
+ name=_job_name(job),
1921
+ threshold=float(_job_threshold(job, suite_options) or 0.7),
1922
+ )
1923
+ payload["kind"] = AGENT_LEARNING_ARTIFACT_EVAL_KIND
1924
+ return payload
1925
+ if command == "eval_task":
1926
+ from fi.alk import evals
1927
+ from fi.alk.cli import AGENT_LEARNING_ARTIFACT_EVAL_KIND
1928
+
1929
+ config_path = _job_optional_path(
1930
+ job,
1931
+ base_dir=base_dir,
1932
+ keys=("config", "eval_config", "agent_report_config"),
1933
+ )
1934
+ config = evals.load_artifact_file(config_path) if config_path else None
1935
+ payload = evals.evaluate_task_evidence_file(
1936
+ path,
1937
+ config=config,
1938
+ name=_job_name(job),
1939
+ threshold=float(_job_threshold(job, suite_options) or 0.7),
1940
+ )
1941
+ payload["kind"] = AGENT_LEARNING_ARTIFACT_EVAL_KIND
1942
+ return payload
1943
+ if command == "redteam":
1944
+ from fi.alk import redteam
1945
+
1946
+ payload = _run_async(
1947
+ redteam.redteam_manifest_file(
1948
+ path,
1949
+ name=_job_name(job),
1950
+ threshold=_job_threshold(job, suite_options),
1951
+ dry_run=_job_dry_run(job, suite_options),
1952
+ )
1953
+ )
1954
+ return payload
1955
+ if command == "optimize":
1956
+ from fi.alk import optimize
1957
+ from fi.alk.cli import AGENT_LEARNING_OPTIMIZATION_KIND
1958
+
1959
+ payload = optimize.optimize_manifest_file(
1960
+ path,
1961
+ name=_job_name(job),
1962
+ threshold=_job_threshold(job, suite_options),
1963
+ max_candidates=_job_max_candidates(job, suite_options),
1964
+ dry_run=_job_dry_run(job, suite_options),
1965
+ )
1966
+ payload["kind"] = AGENT_LEARNING_OPTIMIZATION_KIND
1967
+ return payload
1968
+ if command == "optimize_eval":
1969
+ from fi.alk import optimize
1970
+ from fi.alk.cli import AGENT_LEARNING_EVAL_OPTIMIZATION_KIND
1971
+
1972
+ payload = optimize.optimize_eval_suite_file(
1973
+ path,
1974
+ name=_job_name(job),
1975
+ threshold=_job_threshold(job, suite_options),
1976
+ max_candidates=_job_max_candidates(job, suite_options),
1977
+ dry_run=_job_dry_run(job, suite_options),
1978
+ )
1979
+ payload["kind"] = AGENT_LEARNING_EVAL_OPTIMIZATION_KIND
1980
+ return payload
1981
+ if command == "optimize_suite":
1982
+ from fi.alk import optimize
1983
+ from fi.alk.cli import AGENT_LEARNING_SUITE_OPTIMIZATION_KIND
1984
+
1985
+ payload = optimize.optimize_suite_file(
1986
+ path,
1987
+ name=_job_name(job),
1988
+ threshold=_job_threshold(job, suite_options),
1989
+ max_candidates=_job_max_candidates(job, suite_options),
1990
+ dry_run=_job_dry_run(job, suite_options),
1991
+ )
1992
+ payload["kind"] = AGENT_LEARNING_SUITE_OPTIMIZATION_KIND
1993
+ return payload
1994
+ if command == "baseline":
1995
+ from fi.alk import simulate
1996
+
1997
+ return simulate.create_baseline_file(
1998
+ path,
1999
+ name=_job_name(job),
2000
+ )
2001
+ if command == "compare":
2002
+ from fi.alk import simulate
2003
+
2004
+ return simulate.compare_result_files(
2005
+ _job_compare_baseline_path(job, base_dir=base_dir),
2006
+ path,
2007
+ min_score_delta=_job_float(job, "min_score_delta", "min-score-delta", default=0.0),
2008
+ max_new_findings=_job_int(job, "max_new_findings", "max-new-findings", default=0),
2009
+ max_new_error_findings=_job_int(
2010
+ job,
2011
+ "max_new_error_findings",
2012
+ "max-new-error-findings",
2013
+ default=0,
2014
+ ),
2015
+ min_metric_delta=_job_optional_float(
2016
+ job,
2017
+ "min_metric_delta",
2018
+ "min-metric-delta",
2019
+ ),
2020
+ name=_job_name(job),
2021
+ )
2022
+ if command == "report":
2023
+ from fi.alk import simulate
2024
+
2025
+ return simulate.render_report_file(
2026
+ path,
2027
+ name=_job_name(job),
2028
+ )
2029
+ if command == "promote_to_regression":
2030
+ from fi.alk import simulate
2031
+
2032
+ return simulate.promote_to_regression_file(
2033
+ path,
2034
+ name=_job_name(job),
2035
+ min_level=str(job.get("min_level") or job.get("min-level") or "warning"),
2036
+ max_findings=_job_int(job, "max_findings", "max-findings", default=25),
2037
+ required_env=_as_string_list(job.get("required_env")),
2038
+ )
2039
+ if command == "shrink":
2040
+ from fi.alk import simulate
2041
+
2042
+ return simulate.shrink_attack_evolution_file(
2043
+ path,
2044
+ name=_job_name(job),
2045
+ manifest_name=str(
2046
+ job.get("manifest_name")
2047
+ or job.get("manifest-name")
2048
+ or ""
2049
+ )
2050
+ or None,
2051
+ required_env=_as_string_list(job.get("required_env")),
2052
+ )
2053
+ if command == "replay":
2054
+ from fi.alk import simulate
2055
+
2056
+ return simulate.replay_manifests(
2057
+ _job_replay_manifest_paths(job, base_dir=base_dir),
2058
+ name=_job_name(job),
2059
+ dry_run=_job_dry_run(job, suite_options),
2060
+ fail_fast=bool(suite_options.fail_fast or job.get("fail_fast") or job.get("fail-fast")),
2061
+ )
2062
+ raise SuiteError(f"unsupported suite job command: {command}")
2063
+
2064
+
2065
+ def _write_child_outputs(
2066
+ payload: Mapping[str, Any],
2067
+ *,
2068
+ command: str,
2069
+ job: Mapping[str, Any],
2070
+ path: Path,
2071
+ ) -> list[str]:
2072
+ output_paths = _job_output_paths(job, path.parent)
2073
+ if not any(output_paths.values()):
2074
+ return []
2075
+ render_junit_fn, render_sarif_fn, render_markdown_fn = _child_renderers(command)
2076
+ written: list[str] = []
2077
+ for output_path in output_paths["json"]:
2078
+ output_path.parent.mkdir(parents=True, exist_ok=True)
2079
+ output_path.write_text(
2080
+ json.dumps(payload, indent=2, sort_keys=True, default=str),
2081
+ encoding="utf-8",
2082
+ )
2083
+ written.append(str(output_path))
2084
+ for output_path in output_paths["junit"]:
2085
+ output_path.parent.mkdir(parents=True, exist_ok=True)
2086
+ output_path.write_text(render_junit_fn(payload), encoding="utf-8")
2087
+ written.append(str(output_path))
2088
+ for output_path in output_paths["sarif"]:
2089
+ output_path.parent.mkdir(parents=True, exist_ok=True)
2090
+ output_path.write_text(
2091
+ render_sarif_fn(payload, manifest_path=path),
2092
+ encoding="utf-8",
2093
+ )
2094
+ written.append(str(output_path))
2095
+ for output_path in output_paths["markdown"]:
2096
+ output_path.parent.mkdir(parents=True, exist_ok=True)
2097
+ output_path.write_text(
2098
+ render_markdown_fn(payload, source_path=path),
2099
+ encoding="utf-8",
2100
+ )
2101
+ written.append(str(output_path))
2102
+ return written
2103
+
2104
+
2105
+ def _child_renderers(command: str) -> tuple[Any, Any, Any]:
2106
+ if command == "suite":
2107
+ return render_junit, render_sarif, render_markdown
2108
+ if command == "action_run":
2109
+ from fi.alk import actions, simulate
2110
+
2111
+ def render_action_run_markdown(
2112
+ payload: Mapping[str, Any],
2113
+ *,
2114
+ source_path: Path,
2115
+ ) -> str:
2116
+ return actions.render_action_run_markdown(payload)
2117
+
2118
+ return simulate.render_junit, simulate.render_sarif, render_action_run_markdown
2119
+ if command == "redteam":
2120
+ from fi.alk import redteam
2121
+
2122
+ return redteam.render_junit, redteam.render_sarif, redteam.render_markdown
2123
+ from fi.alk import simulate
2124
+
2125
+ return simulate.render_junit, simulate.render_sarif, simulate.render_markdown
2126
+
2127
+
2128
+ def _suite_child_evidence(
2129
+ job: Mapping[str, Any],
2130
+ child: Mapping[str, Any],
2131
+ *,
2132
+ base_dir: Path,
2133
+ ) -> dict[str, Any]:
2134
+ role = _suite_evidence_role(job, child)
2135
+ exit_code = int(child.get("exit_code", 1))
2136
+ manifest_path = Path(str(child.get("path") or ""))
2137
+ replay_class = str(
2138
+ job.get("replay_class")
2139
+ or job.get("replay")
2140
+ or _as_mapping(job.get("metadata")).get("replay_class")
2141
+ or "r0"
2142
+ )
2143
+ output_digests = _suite_output_digests(
2144
+ child.get("outputs_written"),
2145
+ base_dir=base_dir,
2146
+ )
2147
+ freeze = {
2148
+ "kind": "agent-learning.suite.evidence-freeze.v1",
2149
+ "hash_algorithm": "sha256",
2150
+ "replay_class": replay_class,
2151
+ "manifest": _suite_file_digest(manifest_path),
2152
+ "result_sha256": _suite_json_digest(child.get("result")),
2153
+ "outputs": output_digests,
2154
+ "outputs_sha256": _suite_json_digest(output_digests),
2155
+ }
2156
+ freeze["content_addressed"] = bool(
2157
+ _as_mapping(freeze.get("manifest")).get("sha256")
2158
+ and freeze.get("result_sha256")
2159
+ )
2160
+ reasons: list[str] = []
2161
+ if exit_code != 0:
2162
+ status = "rejected"
2163
+ admitted = False
2164
+ reasons.append("child_failed")
2165
+ elif role in _ADMITTED_EVIDENCE_ROLES:
2166
+ status = "admitted"
2167
+ admitted = True
2168
+ else:
2169
+ status = role if role in _NON_ADMITTED_EVIDENCE_ROLES else "diagnostic"
2170
+ admitted = False
2171
+ reasons.append(f"evidence_role_{status}")
2172
+ if _suite_path_is_fixture(child.get("path")) and status != "rejected":
2173
+ role = "fixture"
2174
+ status = "fixture"
2175
+ admitted = False
2176
+ if "fixture_path" not in reasons:
2177
+ reasons.append("fixture_path")
2178
+ metadata = _as_mapping(job.get("metadata"))
2179
+ claim_scope = (
2180
+ job.get("claim_scope")
2181
+ or job.get("claim")
2182
+ or metadata.get("claim_scope")
2183
+ or ("paper_facing" if admitted else "audit")
2184
+ )
2185
+ return {
2186
+ "kind": "agent-learning.suite.evidence-row.v1",
2187
+ "row_id": str(child.get("id") or job.get("id") or ""),
2188
+ "status": status,
2189
+ "role": role,
2190
+ "admitted": admitted,
2191
+ "reason": reasons,
2192
+ "claim_scope": str(claim_scope),
2193
+ "workload": str(job.get("workload") or job.get("id") or child.get("id") or ""),
2194
+ "driver": str(job.get("driver") or child.get("command") or ""),
2195
+ "command": child.get("command"),
2196
+ "path": child.get("path"),
2197
+ "result_kind": child.get("kind"),
2198
+ "exit_code": exit_code,
2199
+ "provenance": {
2200
+ "job_id": child.get("id") or job.get("id"),
2201
+ "job_name": child.get("name") or job.get("name"),
2202
+ "manifest_path": child.get("path"),
2203
+ "manifest_sha256": _as_mapping(freeze["manifest"]).get("sha256"),
2204
+ "result_sha256": freeze.get("result_sha256"),
2205
+ "outputs_written": list(child.get("outputs_written") or []),
2206
+ "output_digests": output_digests,
2207
+ "outputs_sha256": freeze.get("outputs_sha256"),
2208
+ "replay_class": replay_class,
2209
+ "content_addressed": freeze["content_addressed"],
2210
+ },
2211
+ "freeze": freeze,
2212
+ }
2213
+
2214
+
2215
+ def _suite_evidence_role(
2216
+ job: Mapping[str, Any],
2217
+ child: Mapping[str, Any],
2218
+ ) -> str:
2219
+ metadata = _as_mapping(job.get("metadata"))
2220
+ raw = (
2221
+ job.get("evidence_role")
2222
+ or job.get("evidence_status")
2223
+ or job.get("evidence")
2224
+ or metadata.get("evidence_role")
2225
+ or metadata.get("evidence_status")
2226
+ )
2227
+ role = _suite_key(raw) if raw is not None else ""
2228
+ if role in _ADMITTED_EVIDENCE_ROLES or role in _NON_ADMITTED_EVIDENCE_ROLES:
2229
+ return role
2230
+ if _suite_path_is_fixture(child.get("path") or job.get("path")):
2231
+ return "fixture"
2232
+ return "admitted"
2233
+
2234
+
2235
+ def _suite_path_is_fixture(value: Any) -> bool:
2236
+ text = str(value or "").replace("\\", "/").lower()
2237
+ return "/fixtures/" in text or text.startswith("fixtures/")
2238
+
2239
+
2240
+ def _suite_file_digest(path: str | Path) -> dict[str, Any]:
2241
+ file_path = Path(path).expanduser()
2242
+ exists = file_path.exists()
2243
+ if not exists or not file_path.is_file():
2244
+ return {
2245
+ "path": str(file_path),
2246
+ "exists": exists,
2247
+ "sha256": None,
2248
+ "bytes": 0,
2249
+ }
2250
+ data = file_path.read_bytes()
2251
+ return {
2252
+ "path": str(file_path),
2253
+ "exists": True,
2254
+ "sha256": hashlib.sha256(data).hexdigest(),
2255
+ "bytes": len(data),
2256
+ }
2257
+
2258
+
2259
+ def _suite_json_digest(value: Any) -> str:
2260
+ data = json.dumps(
2261
+ value,
2262
+ sort_keys=True,
2263
+ separators=(",", ":"),
2264
+ default=str,
2265
+ ).encode("utf-8")
2266
+ return hashlib.sha256(data).hexdigest()
2267
+
2268
+
2269
+ def _suite_output_digests(
2270
+ values: Any,
2271
+ *,
2272
+ base_dir: Path,
2273
+ ) -> list[dict[str, Any]]:
2274
+ records: list[dict[str, Any]] = []
2275
+ for value in _as_list(values):
2276
+ path = Path(str(value)).expanduser()
2277
+ if not path.is_absolute():
2278
+ path = (base_dir / path).resolve()
2279
+ records.append(_suite_file_digest(path))
2280
+ return records
2281
+
2282
+
2283
+ def _suite_evidence_admission(
2284
+ children: Sequence[Mapping[str, Any]],
2285
+ ) -> dict[str, Any]:
2286
+ rows = [copy.deepcopy(dict(_as_mapping(child.get("evidence")))) for child in children]
2287
+ rows = [row for row in rows if row]
2288
+ by_status: dict[str, int] = {}
2289
+ by_role: dict[str, int] = {}
2290
+ for row in rows:
2291
+ status = str(row.get("status") or "unknown")
2292
+ role = str(row.get("role") or "unknown")
2293
+ by_status[status] = by_status.get(status, 0) + 1
2294
+ by_role[role] = by_role.get(role, 0) + 1
2295
+ admitted_rows = [row for row in rows if bool(row.get("admitted"))]
2296
+ rejected_rows = [row for row in rows if str(row.get("status") or "") == "rejected"]
2297
+ non_admitted_rows = [row for row in rows if not bool(row.get("admitted"))]
2298
+ frozen_rows = [row for row in rows if _suite_row_content_addressed(row)]
2299
+ admitted_unfrozen_rows = [
2300
+ row for row in admitted_rows if not _suite_row_content_addressed(row)
2301
+ ]
2302
+ return {
2303
+ "kind": "agent-learning.suite.evidence-admission.v1",
2304
+ "admitted_count": len(admitted_rows),
2305
+ "non_admitted_count": len(non_admitted_rows),
2306
+ "rejected_count": len(rejected_rows),
2307
+ "frozen_count": len(frozen_rows),
2308
+ "unfrozen_count": len(rows) - len(frozen_rows),
2309
+ "admitted_frozen_count": len(admitted_rows) - len(admitted_unfrozen_rows),
2310
+ "by_status": dict(sorted(by_status.items())),
2311
+ "by_role": dict(sorted(by_role.items())),
2312
+ "admitted_row_ids": [str(row.get("row_id")) for row in admitted_rows],
2313
+ "non_admitted_row_ids": [str(row.get("row_id")) for row in non_admitted_rows],
2314
+ "admitted_unfrozen_row_ids": [
2315
+ str(row.get("row_id")) for row in admitted_unfrozen_rows
2316
+ ],
2317
+ "rows": rows,
2318
+ }
2319
+
2320
+
2321
+ def _suite_row_content_addressed(row: Mapping[str, Any]) -> bool:
2322
+ freeze = _as_mapping(row.get("freeze"))
2323
+ return bool(freeze.get("content_addressed"))
2324
+
2325
+
2326
+ def _suite_evidence_policy(suite: Mapping[str, Any]) -> dict[str, Any]:
2327
+ raw = (
2328
+ suite.get("evidence_policy")
2329
+ or suite.get("evidence_admission_policy")
2330
+ or suite.get("admission_policy")
2331
+ or {}
2332
+ )
2333
+ if isinstance(raw, Mapping):
2334
+ policy = copy.deepcopy(dict(raw))
2335
+ else:
2336
+ policy = {}
2337
+ min_admitted = policy.get("min_admitted")
2338
+ if min_admitted is None and bool(policy.get("require_admitted")):
2339
+ min_admitted = 1
2340
+ policy["min_admitted"] = int(min_admitted or 0)
2341
+ policy["require_freeze"] = bool(
2342
+ policy.get("require_freeze") or policy.get("require_content_addressed")
2343
+ )
2344
+ return policy
2345
+
2346
+
2347
+ def _suite_optimizer_governance_policy(suite: Mapping[str, Any]) -> dict[str, Any]:
2348
+ raw = (
2349
+ suite.get("optimizer_governance_policy")
2350
+ or suite.get("optimization_governance_policy")
2351
+ or {}
2352
+ )
2353
+ policy = copy.deepcopy(dict(raw)) if isinstance(raw, Mapping) else {}
2354
+ required = bool(
2355
+ policy.get("require_optimizer_governance")
2356
+ or policy.get("required")
2357
+ or policy.get("require_passed")
2358
+ )
2359
+ min_governed = policy.get("min_governed")
2360
+ if min_governed is None and required:
2361
+ min_governed = 1
2362
+ commands = _unique_strings(
2363
+ policy.get("commands")
2364
+ or policy.get("target_commands")
2365
+ or ["optimize"]
2366
+ )
2367
+ policy["require_optimizer_governance"] = required
2368
+ policy["require_passed"] = bool(policy.get("require_passed") or required)
2369
+ policy["fail_on_warning"] = bool(policy.get("fail_on_warning"))
2370
+ policy["min_governed"] = int(min_governed or 0)
2371
+ policy["commands"] = commands or ["optimize"]
2372
+ return policy
2373
+
2374
+
2375
+ def _suite_optimizer_governance(
2376
+ children: Sequence[Mapping[str, Any]],
2377
+ policy: Mapping[str, Any],
2378
+ ) -> dict[str, Any]:
2379
+ target_commands = {
2380
+ _normalize_command(command)
2381
+ for command in _as_list(policy.get("commands"))
2382
+ if command
2383
+ }
2384
+ rows = [
2385
+ _suite_optimizer_governance_row(child)
2386
+ for child in children
2387
+ if _suite_optimizer_governance_targets_child(child, target_commands)
2388
+ ]
2389
+ governed_rows = [row for row in rows if bool(row.get("governance_present"))]
2390
+ failed_rows = [
2391
+ row
2392
+ for row in governed_rows
2393
+ if row.get("governance_status") != "passed" or row.get("passed") is False
2394
+ ]
2395
+ missing_rows = [row for row in rows if not bool(row.get("governance_present"))]
2396
+ warning_rows = [
2397
+ row
2398
+ for row in governed_rows
2399
+ if _as_list(row.get("warning_check_ids"))
2400
+ ]
2401
+ return {
2402
+ "kind": "agent-learning.suite.optimizer-governance.v1",
2403
+ "status": "failed" if failed_rows or missing_rows else "passed",
2404
+ "policy": copy.deepcopy(dict(policy)),
2405
+ "target_count": len(rows),
2406
+ "governed_count": len(governed_rows),
2407
+ "passed_count": len(governed_rows) - len(failed_rows),
2408
+ "failed_count": len(failed_rows),
2409
+ "missing_count": len(missing_rows),
2410
+ "warning_count": len(warning_rows),
2411
+ "target_child_ids": [str(row.get("child_id")) for row in rows],
2412
+ "governed_child_ids": [str(row.get("child_id")) for row in governed_rows],
2413
+ "failed_child_ids": [str(row.get("child_id")) for row in failed_rows],
2414
+ "missing_child_ids": [str(row.get("child_id")) for row in missing_rows],
2415
+ "warning_child_ids": [str(row.get("child_id")) for row in warning_rows],
2416
+ "rows": rows,
2417
+ }
2418
+
2419
+
2420
+ def _suite_optimizer_governance_targets_child(
2421
+ child: Mapping[str, Any],
2422
+ target_commands: set[str],
2423
+ ) -> bool:
2424
+ result = _as_mapping(child.get("result"))
2425
+ if _as_mapping(result.get("optimization_governance")):
2426
+ return True
2427
+ command = _normalize_command(child.get("command") or "")
2428
+ if command in target_commands:
2429
+ return True
2430
+ return False
2431
+
2432
+
2433
+ def _suite_optimizer_governance_row(child: Mapping[str, Any]) -> dict[str, Any]:
2434
+ result = _as_mapping(child.get("result"))
2435
+ governance = _as_mapping(result.get("optimization_governance"))
2436
+ if not governance:
2437
+ governance = _as_mapping(_as_mapping(result.get("optimization")).get("governance"))
2438
+ evidence = _as_mapping(governance.get("evidence"))
2439
+ return {
2440
+ "kind": "agent-learning.suite.optimizer-governance-row.v1",
2441
+ "child_id": child.get("id"),
2442
+ "command": child.get("command"),
2443
+ "path": child.get("path"),
2444
+ "result_kind": child.get("kind"),
2445
+ "child_status": child.get("status"),
2446
+ "child_exit_code": int(child.get("exit_code", 1)),
2447
+ "governance_present": bool(governance),
2448
+ "governance_kind": governance.get("kind"),
2449
+ "governance_status": governance.get("status") if governance else "missing",
2450
+ "passed": bool(governance.get("passed")) if governance else False,
2451
+ "selected_candidate_id": governance.get("selected_candidate_id"),
2452
+ "selected_rank": governance.get("selected_rank"),
2453
+ "check_count": int(governance.get("check_count") or 0),
2454
+ "failed_check_ids": [
2455
+ str(item) for item in _as_list(governance.get("failed_check_ids"))
2456
+ ],
2457
+ "warning_check_ids": [
2458
+ str(item) for item in _as_list(governance.get("warning_check_ids"))
2459
+ ],
2460
+ "candidate_count": int(evidence.get("candidate_count") or 0),
2461
+ "content_addressed_count": int(
2462
+ evidence.get("content_addressed_count") or 0
2463
+ ),
2464
+ "metric_count": int(evidence.get("metric_count") or 0),
2465
+ "patch_path_count": int(evidence.get("patch_path_count") or 0),
2466
+ }
2467
+
2468
+
2469
+ def _suite_optimizer_governance_findings(
2470
+ optimizer_governance: Mapping[str, Any],
2471
+ policy: Mapping[str, Any],
2472
+ ) -> list[dict[str, Any]]:
2473
+ findings: list[dict[str, Any]] = []
2474
+ min_governed = int(policy.get("min_governed") or 0)
2475
+ governed_count = int(optimizer_governance.get("governed_count") or 0)
2476
+ if min_governed > governed_count:
2477
+ findings.append({
2478
+ "type": "suite_optimizer_governance_missing",
2479
+ "level": "error",
2480
+ "reason": (
2481
+ f"Suite optimizer governance gate requires at least {min_governed} "
2482
+ f"governed optimizer child row(s), but only {governed_count} "
2483
+ "were found."
2484
+ ),
2485
+ "min_governed": min_governed,
2486
+ "governed_count": governed_count,
2487
+ "missing_child_ids": list(
2488
+ optimizer_governance.get("missing_child_ids") or []
2489
+ ),
2490
+ })
2491
+ if bool(policy.get("require_passed")):
2492
+ failed_child_ids = list(optimizer_governance.get("failed_child_ids") or [])
2493
+ missing_child_ids = list(optimizer_governance.get("missing_child_ids") or [])
2494
+ blocked_child_ids = sorted(
2495
+ {str(item) for item in [*failed_child_ids, *missing_child_ids]}
2496
+ )
2497
+ if blocked_child_ids:
2498
+ findings.append({
2499
+ "type": "suite_optimizer_governance_failed",
2500
+ "level": "error",
2501
+ "reason": (
2502
+ "Suite optimizer governance gate requires passed governance "
2503
+ f"for optimizer children, but {len(blocked_child_ids)} child "
2504
+ "row(s) are missing or failed."
2505
+ ),
2506
+ "failed_child_ids": failed_child_ids,
2507
+ "missing_child_ids": missing_child_ids,
2508
+ })
2509
+ if bool(policy.get("fail_on_warning")):
2510
+ warning_child_ids = list(optimizer_governance.get("warning_child_ids") or [])
2511
+ if warning_child_ids:
2512
+ findings.append({
2513
+ "type": "suite_optimizer_governance_warning",
2514
+ "level": "error",
2515
+ "reason": (
2516
+ "Suite optimizer governance gate is configured to fail on "
2517
+ f"warnings, and {len(warning_child_ids)} child row(s) have "
2518
+ "governance warnings."
2519
+ ),
2520
+ "warning_child_ids": warning_child_ids,
2521
+ })
2522
+ return findings
2523
+
2524
+
2525
+ def _suite_evidence_findings(
2526
+ admission: Mapping[str, Any],
2527
+ policy: Mapping[str, Any],
2528
+ ) -> list[dict[str, Any]]:
2529
+ min_admitted = int(policy.get("min_admitted") or 0)
2530
+ admitted_count = int(admission.get("admitted_count") or 0)
2531
+ findings: list[dict[str, Any]] = []
2532
+ if min_admitted > admitted_count:
2533
+ findings.append({
2534
+ "type": "suite_evidence_admission_missing",
2535
+ "level": "error",
2536
+ "reason": (
2537
+ f"Suite evidence gate requires at least {min_admitted} admitted "
2538
+ f"row(s), but only {admitted_count} were admitted."
2539
+ ),
2540
+ "admitted_count": admitted_count,
2541
+ "min_admitted": min_admitted,
2542
+ })
2543
+ if bool(policy.get("require_freeze")):
2544
+ missing = [
2545
+ str(row_id)
2546
+ for row_id in _as_list(admission.get("admitted_unfrozen_row_ids"))
2547
+ ]
2548
+ if missing:
2549
+ findings.append({
2550
+ "type": "suite_evidence_freeze_missing",
2551
+ "level": "error",
2552
+ "reason": (
2553
+ "Suite evidence gate requires content-addressed admitted "
2554
+ f"rows, but {len(missing)} admitted row(s) are missing "
2555
+ "manifest/result digests."
2556
+ ),
2557
+ "missing": missing,
2558
+ })
2559
+ return findings
2560
+
2561
+
2562
+ def _suite_result(
2563
+ *,
2564
+ suite: Mapping[str, Any],
2565
+ suite_path: Path,
2566
+ children: Sequence[Mapping[str, Any]],
2567
+ name: Optional[str],
2568
+ dry_run: bool,
2569
+ fail_fast: bool,
2570
+ duration_seconds: float,
2571
+ ) -> dict[str, Any]:
2572
+ job_count = len(_suite_jobs(suite))
2573
+ passed = [child for child in children if int(child.get("exit_code", 1)) == 0]
2574
+ failed = [child for child in children if int(child.get("exit_code", 1)) != 0]
2575
+ score = round(len(passed) / job_count, 4) if job_count else 0.0
2576
+ command_counts: dict[str, int] = {}
2577
+ for child in children:
2578
+ command = str(child.get("command") or "unknown")
2579
+ command_counts[command] = command_counts.get(command, 0) + 1
2580
+ capabilities = _suite_capability_summary(children)
2581
+ required_capabilities = _suite_required_capabilities(suite)
2582
+ missing_capabilities = _missing_required_capabilities(
2583
+ required_capabilities,
2584
+ capabilities,
2585
+ )
2586
+ capability_findings = _suite_capability_findings(missing_capabilities)
2587
+ framework_coverage = _suite_framework_coverage(
2588
+ children,
2589
+ required_frameworks=required_capabilities.get("frameworks", []),
2590
+ )
2591
+ framework_findings = _suite_framework_findings(framework_coverage)
2592
+ evidence_admission = _suite_evidence_admission(children)
2593
+ evidence_policy = _suite_evidence_policy(suite)
2594
+ evidence_findings = _suite_evidence_findings(
2595
+ evidence_admission,
2596
+ evidence_policy,
2597
+ )
2598
+ optimizer_governance_policy = _suite_optimizer_governance_policy(suite)
2599
+ optimizer_governance = _suite_optimizer_governance(
2600
+ children,
2601
+ optimizer_governance_policy,
2602
+ )
2603
+ optimizer_governance_findings = _suite_optimizer_governance_findings(
2604
+ optimizer_governance,
2605
+ optimizer_governance_policy,
2606
+ )
2607
+ suite_findings = [
2608
+ *capability_findings,
2609
+ *framework_findings,
2610
+ *evidence_findings,
2611
+ *optimizer_governance_findings,
2612
+ *_suite_findings(children),
2613
+ ]
2614
+ suite_passed = (
2615
+ len(failed) == 0
2616
+ and len(children) == job_count
2617
+ and not capability_findings
2618
+ and not framework_findings
2619
+ and not evidence_findings
2620
+ and not optimizer_governance_findings
2621
+ )
2622
+ trust_certificate = _suite_trust_certificate(
2623
+ suite=suite,
2624
+ suite_path=suite_path,
2625
+ children=children,
2626
+ capabilities=capabilities,
2627
+ framework_coverage=framework_coverage,
2628
+ evidence_admission=evidence_admission,
2629
+ optimizer_governance=optimizer_governance,
2630
+ missing_capabilities=missing_capabilities,
2631
+ suite_passed=suite_passed,
2632
+ job_count=job_count,
2633
+ executed_count=len(children),
2634
+ passed_count=len(passed),
2635
+ failed_count=len(failed),
2636
+ score=score,
2637
+ )
2638
+ return {
2639
+ "kind": AGENT_LEARNING_SUITE_KIND,
2640
+ "version": AGENT_LEARNING_SUITE_KIND,
2641
+ "name": str(name or suite.get("name") or suite_path.stem),
2642
+ "status": "passed" if suite_passed else "failed",
2643
+ "exit_code": 0 if suite_passed else 1,
2644
+ "dry_run": dry_run,
2645
+ "fail_fast": fail_fast,
2646
+ "summary": {
2647
+ "job_count": job_count,
2648
+ "executed_count": len(children),
2649
+ "passed_count": len(passed),
2650
+ "failed_count": len(failed),
2651
+ "skipped_count": max(job_count - len(children), 0),
2652
+ "score": score,
2653
+ "trust_certificate_verdict": trust_certificate["verdict"],
2654
+ "trust_certificate_assurance_level": trust_certificate[
2655
+ "assurance_level"
2656
+ ],
2657
+ "trust_certificate_promotion_ready": trust_certificate[
2658
+ "promotion_ready"
2659
+ ],
2660
+ "trust_certificate_failed_gate_count": len(
2661
+ trust_certificate["failed_gate_ids"]
2662
+ ),
2663
+ "trust_certificate_conditional_gate_count": len(
2664
+ trust_certificate["conditional_gate_ids"]
2665
+ ),
2666
+ "commands": command_counts,
2667
+ "capabilities": capabilities,
2668
+ "required_capabilities": required_capabilities,
2669
+ "missing_required_capabilities": missing_capabilities,
2670
+ "capability_gate_passed": not capability_findings,
2671
+ "framework_coverage_passed": not framework_findings,
2672
+ "observed_framework_count": framework_coverage["observed_count"],
2673
+ "required_framework_count": framework_coverage["required_count"],
2674
+ "missing_framework_count": framework_coverage["missing_count"],
2675
+ "adapter_conformance_failed_count": framework_coverage[
2676
+ "adapter_conformance_failed_count"
2677
+ ],
2678
+ "framework_coverage": {
2679
+ key: value
2680
+ for key, value in framework_coverage.items()
2681
+ if key != "rows"
2682
+ },
2683
+ "evidence_gate_passed": not evidence_findings,
2684
+ "optimizer_governance_gate_passed": not optimizer_governance_findings,
2685
+ "optimizer_governance_policy": optimizer_governance_policy,
2686
+ "optimizer_governance_target_count": optimizer_governance[
2687
+ "target_count"
2688
+ ],
2689
+ "optimizer_governance_governed_count": optimizer_governance[
2690
+ "governed_count"
2691
+ ],
2692
+ "optimizer_governance_passed_count": optimizer_governance[
2693
+ "passed_count"
2694
+ ],
2695
+ "optimizer_governance_failed_count": optimizer_governance[
2696
+ "failed_count"
2697
+ ],
2698
+ "optimizer_governance_missing_count": optimizer_governance[
2699
+ "missing_count"
2700
+ ],
2701
+ "optimizer_governance_warning_count": optimizer_governance[
2702
+ "warning_count"
2703
+ ],
2704
+ "admitted_evidence_count": evidence_admission["admitted_count"],
2705
+ "non_admitted_evidence_count": evidence_admission[
2706
+ "non_admitted_count"
2707
+ ],
2708
+ "rejected_evidence_count": evidence_admission["rejected_count"],
2709
+ "frozen_evidence_count": evidence_admission["frozen_count"],
2710
+ "unfrozen_evidence_count": evidence_admission["unfrozen_count"],
2711
+ "admitted_frozen_evidence_count": evidence_admission[
2712
+ "admitted_frozen_count"
2713
+ ],
2714
+ "evidence_admission": {
2715
+ key: value
2716
+ for key, value in evidence_admission.items()
2717
+ if key != "rows"
2718
+ },
2719
+ },
2720
+ "framework_coverage": framework_coverage,
2721
+ "evidence_admission": evidence_admission,
2722
+ "optimizer_governance": optimizer_governance,
2723
+ "trust_certificate": trust_certificate,
2724
+ "children": list(children),
2725
+ "jobs": list(children),
2726
+ "findings": suite_findings,
2727
+ "duration_seconds": duration_seconds,
2728
+ }
2729
+
2730
+
2731
+ def _suite_descriptor(suite: Mapping[str, Any]) -> dict[str, Any]:
2732
+ return {
2733
+ "version": suite.get("version") or AGENT_LEARNING_SUITE_KIND,
2734
+ "name": suite.get("name"),
2735
+ "job_count": len(_suite_jobs(suite)),
2736
+ "jobs": [
2737
+ {
2738
+ "id": job.get("id"),
2739
+ "command": job.get("command"),
2740
+ "path": job.get("path"),
2741
+ }
2742
+ for job in _suite_jobs(suite)
2743
+ ],
2744
+ "required_capabilities": _suite_required_capabilities(suite),
2745
+ }
2746
+
2747
+
2748
+ def _suite_trust_certificate(
2749
+ *,
2750
+ suite: Mapping[str, Any],
2751
+ suite_path: Path,
2752
+ children: Sequence[Mapping[str, Any]],
2753
+ capabilities: Mapping[str, Sequence[str]],
2754
+ framework_coverage: Mapping[str, Any],
2755
+ evidence_admission: Mapping[str, Any],
2756
+ optimizer_governance: Mapping[str, Any],
2757
+ missing_capabilities: Mapping[str, Sequence[str]],
2758
+ suite_passed: bool,
2759
+ job_count: int,
2760
+ executed_count: int,
2761
+ passed_count: int,
2762
+ failed_count: int,
2763
+ score: float,
2764
+ ) -> dict[str, Any]:
2765
+ coverage = _suite_trinity_coverage(capabilities)
2766
+ admitted_count = int(evidence_admission.get("admitted_count") or 0)
2767
+ admitted_frozen_count = int(evidence_admission.get("admitted_frozen_count") or 0)
2768
+ governed_count = int(optimizer_governance.get("governed_count") or 0)
2769
+ optimizer_failed_count = int(optimizer_governance.get("failed_count") or 0)
2770
+ optimizer_missing_count = int(optimizer_governance.get("missing_count") or 0)
2771
+ gates = [
2772
+ _trust_gate(
2773
+ "execution",
2774
+ passed=failed_count == 0 and executed_count == job_count and suite_passed,
2775
+ required=True,
2776
+ reason="all declared suite jobs executed and exited successfully",
2777
+ evidence={
2778
+ "job_count": job_count,
2779
+ "executed_count": executed_count,
2780
+ "passed_count": passed_count,
2781
+ "failed_count": failed_count,
2782
+ "score": score,
2783
+ },
2784
+ ),
2785
+ _trust_gate(
2786
+ "capability_gate",
2787
+ passed=not missing_capabilities,
2788
+ required=True,
2789
+ reason="declared required capabilities were observed",
2790
+ evidence={"missing_required_capabilities": dict(missing_capabilities)},
2791
+ ),
2792
+ _trust_gate(
2793
+ "framework_coverage",
2794
+ passed=int(framework_coverage.get("missing_count") or 0) == 0
2795
+ and int(framework_coverage.get("adapter_conformance_failed_count") or 0)
2796
+ == 0,
2797
+ required=True,
2798
+ reason="required framework coverage and adapter conformance passed",
2799
+ evidence={
2800
+ "observed_count": framework_coverage.get("observed_count"),
2801
+ "required_count": framework_coverage.get("required_count"),
2802
+ "missing_count": framework_coverage.get("missing_count"),
2803
+ "adapter_conformance_failed_count": framework_coverage.get(
2804
+ "adapter_conformance_failed_count"
2805
+ ),
2806
+ },
2807
+ ),
2808
+ _trust_gate(
2809
+ "evidence_admission",
2810
+ passed=admitted_count > 0
2811
+ and int(evidence_admission.get("rejected_count") or 0) == 0,
2812
+ required=False,
2813
+ reason="at least one child artifact is admitted evidence",
2814
+ evidence={
2815
+ "admitted_count": admitted_count,
2816
+ "rejected_count": evidence_admission.get("rejected_count"),
2817
+ "by_status": evidence_admission.get("by_status"),
2818
+ },
2819
+ ),
2820
+ _trust_gate(
2821
+ "evidence_freeze",
2822
+ passed=admitted_count > 0 and admitted_frozen_count == admitted_count,
2823
+ required=False,
2824
+ reason="admitted evidence rows are content-addressed",
2825
+ evidence={
2826
+ "admitted_count": admitted_count,
2827
+ "admitted_frozen_count": admitted_frozen_count,
2828
+ },
2829
+ ),
2830
+ _trust_gate(
2831
+ "optimizer_governance",
2832
+ passed=governed_count > 0
2833
+ and optimizer_failed_count == 0
2834
+ and optimizer_missing_count == 0,
2835
+ required=False,
2836
+ reason="optimizer children expose passed governance verdicts",
2837
+ evidence={
2838
+ "target_count": optimizer_governance.get("target_count"),
2839
+ "governed_count": governed_count,
2840
+ "failed_count": optimizer_failed_count,
2841
+ "missing_count": optimizer_missing_count,
2842
+ "warning_count": optimizer_governance.get("warning_count"),
2843
+ },
2844
+ ),
2845
+ _trust_gate(
2846
+ "trinity_coverage",
2847
+ passed=all(coverage.values()),
2848
+ required=False,
2849
+ reason="suite covers simulation, evaluation, red-team, and optimization",
2850
+ evidence=coverage,
2851
+ ),
2852
+ ]
2853
+ failed_gate_ids = [
2854
+ gate["id"] for gate in gates if gate["required"] and not gate["passed"]
2855
+ ]
2856
+ conditional_gate_ids = [
2857
+ gate["id"] for gate in gates if not gate["required"] and not gate["passed"]
2858
+ ]
2859
+ if not suite_passed or failed_gate_ids:
2860
+ verdict = "rejected"
2861
+ elif conditional_gate_ids:
2862
+ verdict = "conditional"
2863
+ else:
2864
+ verdict = "approved"
2865
+ return {
2866
+ "kind": "agent-learning.suite.trust-certificate.v1",
2867
+ "verdict": verdict,
2868
+ "promotion_ready": verdict == "approved",
2869
+ "assurance_level": _suite_assurance_level(verdict, coverage, governed_count),
2870
+ "subject": {
2871
+ "suite_name": str(suite.get("name") or suite_path.stem),
2872
+ "suite_path": str(suite_path),
2873
+ "suite_version": suite.get("version") or AGENT_LEARNING_SUITE_KIND,
2874
+ "job_count": job_count,
2875
+ },
2876
+ "coverage": coverage,
2877
+ "evidence": {
2878
+ "admitted_count": admitted_count,
2879
+ "admitted_frozen_count": admitted_frozen_count,
2880
+ "optimizer_governed_count": governed_count,
2881
+ "optimizer_failed_count": optimizer_failed_count,
2882
+ "optimizer_missing_count": optimizer_missing_count,
2883
+ "framework_observed_count": framework_coverage.get("observed_count"),
2884
+ "framework_missing_count": framework_coverage.get("missing_count"),
2885
+ },
2886
+ "failed_gate_ids": failed_gate_ids,
2887
+ "conditional_gate_ids": conditional_gate_ids,
2888
+ "reason": _suite_trust_reason(verdict, failed_gate_ids, conditional_gate_ids),
2889
+ "gates": gates,
2890
+ "child_ids": [str(child.get("id") or "") for child in children],
2891
+ }
2892
+
2893
+
2894
+ def _suite_trinity_coverage(capabilities: Mapping[str, Sequence[str]]) -> dict[str, bool]:
2895
+ commands = {_suite_key(command) for command in _as_list(capabilities.get("commands"))}
2896
+ result_kinds = {
2897
+ str(item)
2898
+ for item in _as_list(capabilities.get("result_kinds"))
2899
+ if str(item)
2900
+ }
2901
+ return {
2902
+ "simulation": "run" in commands or "agent-learning.run.v1" in result_kinds,
2903
+ "evaluation": bool(
2904
+ commands & {"eval", "eval_artifact", "eval_task", "optimize_eval"}
2905
+ )
2906
+ or "agent-learning.eval.v1" in result_kinds,
2907
+ "redteam": "redteam" in commands or "agent-learning.redteam.v1" in result_kinds,
2908
+ "optimization": bool(commands & {"optimize", "optimize_eval", "optimize_suite"})
2909
+ or "agent-learning.optimization.v1" in result_kinds
2910
+ or "agent-learning.suite-optimization.v1" in result_kinds,
2911
+ }
2912
+
2913
+
2914
+ def _suite_assurance_level(
2915
+ verdict: str,
2916
+ coverage: Mapping[str, bool],
2917
+ governed_count: int,
2918
+ ) -> str:
2919
+ if verdict == "rejected":
2920
+ return "rejected"
2921
+ if all(coverage.values()) and governed_count > 0:
2922
+ return "l3_trinity_governed"
2923
+ if coverage.get("simulation") and coverage.get("evaluation"):
2924
+ return "l2_evaluated_simulation"
2925
+ return "l1_partial_evidence"
2926
+
2927
+
2928
+ def _suite_trust_reason(
2929
+ verdict: str,
2930
+ failed_gate_ids: Sequence[str],
2931
+ conditional_gate_ids: Sequence[str],
2932
+ ) -> str:
2933
+ if verdict == "approved":
2934
+ return (
2935
+ "Approved: execution, evidence, framework coverage, red-team, "
2936
+ "simulation, evaluation, optimization, and optimizer governance closed."
2937
+ )
2938
+ if verdict == "rejected":
2939
+ return (
2940
+ "Rejected: required suite gates failed"
2941
+ + (f" ({', '.join(failed_gate_ids)})." if failed_gate_ids else ".")
2942
+ )
2943
+ return (
2944
+ "Conditional: required gates passed but advisory deployment evidence is "
2945
+ f"incomplete ({', '.join(conditional_gate_ids)})."
2946
+ )
2947
+
2948
+
2949
+ def _trust_gate(
2950
+ gate_id: str,
2951
+ *,
2952
+ passed: bool,
2953
+ required: bool,
2954
+ reason: str,
2955
+ evidence: Mapping[str, Any],
2956
+ ) -> dict[str, Any]:
2957
+ return {
2958
+ "id": gate_id,
2959
+ "status": "passed" if passed else "failed" if required else "conditional",
2960
+ "passed": passed,
2961
+ "required": required,
2962
+ "reason": reason,
2963
+ "evidence": copy.deepcopy(dict(evidence)),
2964
+ }
2965
+
2966
+
2967
+ def _suite_framework_coverage(
2968
+ children: Sequence[Mapping[str, Any]],
2969
+ *,
2970
+ required_frameworks: Sequence[str],
2971
+ ) -> dict[str, Any]:
2972
+ rows: list[dict[str, Any]] = []
2973
+ for child in children:
2974
+ rows.extend(_suite_framework_rows_for_child(_as_mapping(child)))
2975
+ observed = sorted(
2976
+ {
2977
+ _suite_key(row.get("framework"))
2978
+ for row in rows
2979
+ if _suite_key(row.get("framework"))
2980
+ }
2981
+ )
2982
+ required = sorted(
2983
+ {
2984
+ _suite_key(item)
2985
+ for item in _as_list(required_frameworks)
2986
+ if _suite_key(item)
2987
+ }
2988
+ )
2989
+ missing = sorted(set(required) - set(observed))
2990
+ adapter_failures = [
2991
+ row
2992
+ for row in rows
2993
+ if row.get("adapter_conformance_passed") is False
2994
+ ]
2995
+ methods: dict[str, set[str]] = {}
2996
+ input_modes: dict[str, set[str]] = {}
2997
+ modalities: dict[str, set[str]] = {}
2998
+ for row in rows:
2999
+ framework = _suite_key(row.get("framework"))
3000
+ if not framework:
3001
+ continue
3002
+ methods.setdefault(framework, set()).update(
3003
+ _suite_key(item)
3004
+ for item in _as_list(row.get("methods"))
3005
+ if _suite_key(item)
3006
+ )
3007
+ input_modes.setdefault(framework, set()).update(
3008
+ _suite_key(item)
3009
+ for item in _as_list(row.get("input_modes"))
3010
+ if _suite_key(item)
3011
+ )
3012
+ modality = _suite_key(row.get("modality"))
3013
+ if modality:
3014
+ modalities.setdefault(framework, set()).add(modality)
3015
+ return {
3016
+ "kind": "agent-learning.suite.framework-coverage.v1",
3017
+ "observed_frameworks": observed,
3018
+ "required_frameworks": required,
3019
+ "missing_required_frameworks": missing,
3020
+ "observed_count": len(observed),
3021
+ "required_count": len(required),
3022
+ "missing_count": len(missing),
3023
+ "adapter_conformance_failed_count": len(adapter_failures),
3024
+ "adapter_conformance_failed_child_ids": [
3025
+ str(row.get("child_id")) for row in adapter_failures
3026
+ ],
3027
+ "methods_by_framework": {
3028
+ key: sorted(values) for key, values in sorted(methods.items())
3029
+ },
3030
+ "input_modes_by_framework": {
3031
+ key: sorted(values) for key, values in sorted(input_modes.items())
3032
+ },
3033
+ "modalities_by_framework": {
3034
+ key: sorted(values) for key, values in sorted(modalities.items())
3035
+ },
3036
+ "rows": rows,
3037
+ }
3038
+
3039
+
3040
+ def _suite_framework_findings(
3041
+ coverage: Mapping[str, Any],
3042
+ ) -> list[dict[str, Any]]:
3043
+ findings: list[dict[str, Any]] = []
3044
+ missing = [
3045
+ _suite_key(item)
3046
+ for item in _as_list(coverage.get("missing_required_frameworks"))
3047
+ if _suite_key(item)
3048
+ ]
3049
+ if missing:
3050
+ findings.append(
3051
+ {
3052
+ "type": "suite_framework_coverage_missing",
3053
+ "level": "error",
3054
+ "reason": (
3055
+ "Suite framework coverage is missing required framework(s): "
3056
+ f"{', '.join(sorted(missing))}."
3057
+ ),
3058
+ "missing": sorted(missing),
3059
+ }
3060
+ )
3061
+ failed = [
3062
+ str(item)
3063
+ for item in _as_list(coverage.get("adapter_conformance_failed_child_ids"))
3064
+ if str(item)
3065
+ ]
3066
+ if failed:
3067
+ findings.append(
3068
+ {
3069
+ "type": "suite_framework_adapter_conformance_failed",
3070
+ "level": "error",
3071
+ "reason": (
3072
+ "Suite framework coverage found adapter conformance failures "
3073
+ f"in {len(failed)} child row(s)."
3074
+ ),
3075
+ "failed_child_ids": failed,
3076
+ }
3077
+ )
3078
+ return findings
3079
+
3080
+
3081
+ def _suite_framework_rows_for_child(child: Mapping[str, Any]) -> list[dict[str, Any]]:
3082
+ rows: list[dict[str, Any]] = []
3083
+ result = _as_mapping(child.get("result"))
3084
+ for nested in _as_list(result.get("children") or result.get("jobs")):
3085
+ nested_child = _as_mapping(nested)
3086
+ if nested_child:
3087
+ rows.extend(_suite_framework_rows_for_child(nested_child))
3088
+ for state in _suite_framework_environment_states(result):
3089
+ row = _suite_framework_row_from_state(child, state)
3090
+ if row:
3091
+ rows.append(row)
3092
+ return rows
3093
+
3094
+
3095
+ def _suite_framework_environment_states(
3096
+ result: Mapping[str, Any],
3097
+ ) -> list[dict[str, Any]]:
3098
+ states: list[dict[str, Any]] = []
3099
+ for report in (
3100
+ _as_mapping(result.get("report")),
3101
+ _as_mapping(_as_mapping(result.get("evaluation")).get("report")),
3102
+ ):
3103
+ for case in _as_list(report.get("results")):
3104
+ metadata = _as_mapping(_as_mapping(case).get("metadata"))
3105
+ state = _as_mapping(metadata.get("environment_state"))
3106
+ if state:
3107
+ states.append(state)
3108
+ return states
3109
+
3110
+
3111
+ def _suite_framework_row_from_state(
3112
+ child: Mapping[str, Any],
3113
+ state: Mapping[str, Any],
3114
+ ) -> dict[str, Any] | None:
3115
+ runtime = _as_mapping(state.get("framework_runtime"))
3116
+ trace = _as_mapping(state.get("framework_trace"))
3117
+ capability = _as_mapping(state.get("framework_capability_matrix"))
3118
+ framework = (
3119
+ runtime.get("framework")
3120
+ or trace.get("framework")
3121
+ or capability.get("framework")
3122
+ )
3123
+ framework_key = _suite_key(framework)
3124
+ if not framework_key:
3125
+ return None
3126
+ runtime_summary = _as_mapping(runtime.get("summary"))
3127
+ trace_spans = [
3128
+ _as_mapping(span)
3129
+ for span in _as_list(trace.get("spans"))
3130
+ if _as_mapping(span)
3131
+ ]
3132
+ trace_signals = sorted(
3133
+ {
3134
+ _suite_key(signal)
3135
+ for span in trace_spans
3136
+ for signal in _as_list(span.get("signals"))
3137
+ if _suite_key(signal)
3138
+ }
3139
+ )
3140
+ conformance = _as_mapping(trace.get("adapter_conformance"))
3141
+ conformance_passed = (
3142
+ bool(conformance.get("passed")) if conformance else None
3143
+ )
3144
+ return {
3145
+ "kind": "agent-learning.suite.framework-coverage-row.v1",
3146
+ "child_id": child.get("id"),
3147
+ "child_name": child.get("name"),
3148
+ "command": child.get("command"),
3149
+ "result_kind": child.get("kind"),
3150
+ "framework": framework_key,
3151
+ "modality": _suite_key(runtime.get("modality") or trace.get("modality")),
3152
+ "methods": sorted(
3153
+ {
3154
+ _suite_key(item)
3155
+ for item in _as_list(runtime_summary.get("methods"))
3156
+ if _suite_key(item)
3157
+ }
3158
+ ),
3159
+ "input_modes": sorted(
3160
+ {
3161
+ _suite_key(item)
3162
+ for item in _as_list(runtime_summary.get("input_modes"))
3163
+ if _suite_key(item)
3164
+ }
3165
+ ),
3166
+ "tool_call_count": int(runtime_summary.get("tool_call_count") or 0),
3167
+ "trace_span_count": len(trace_spans),
3168
+ "trace_signals": trace_signals,
3169
+ "adapter_conformance_passed": conformance_passed,
3170
+ }
3171
+
3172
+
3173
+ def _suite_job_command_counts(suite: Mapping[str, Any]) -> dict[str, int]:
3174
+ counts: dict[str, int] = {}
3175
+ for job in _suite_jobs(suite):
3176
+ command = str(job.get("command") or "unknown")
3177
+ counts[command] = counts.get(command, 0) + 1
3178
+ return counts
3179
+
3180
+
3181
+ def _artifact_action_plan_card(result: Mapping[str, Any]) -> dict[str, Any] | None:
3182
+ optimization = _as_mapping(result.get("optimization"))
3183
+ history = [
3184
+ _as_mapping(item)
3185
+ for item in _as_list(optimization.get("history"))
3186
+ if _as_mapping(item)
3187
+ ]
3188
+ candidate_records = [
3189
+ record
3190
+ for item in history
3191
+ for record in _artifact_action_candidate_records(item)
3192
+ ]
3193
+ if not candidate_records:
3194
+ return None
3195
+ selected_action_id = _artifact_action_selected_id(optimization, candidate_records)
3196
+ for record in candidate_records:
3197
+ record["selected"] = bool(record.get("action_id") == selected_action_id)
3198
+ selected = next(
3199
+ (
3200
+ record
3201
+ for record in candidate_records
3202
+ if record.get("action_id") == selected_action_id
3203
+ ),
3204
+ max(candidate_records, key=lambda record: float(record.get("score") or 0.0)),
3205
+ )
3206
+ return {
3207
+ "kind": "artifact_action_plan",
3208
+ "status": "selected" if selected_action_id else "observed",
3209
+ "source": "agent_learning_suite_optimization",
3210
+ "selected_action_id": selected.get("action_id"),
3211
+ "selected_candidate_id": selected.get("candidate_id"),
3212
+ "selected_score": selected.get("score"),
3213
+ "selection_reason": _artifact_action_selection_reason(selected),
3214
+ "candidate_count": len(candidate_records),
3215
+ "candidate_score_lineage": candidate_records,
3216
+ "search_paths": _as_string_list(result.get("summary", {}).get("search_paths")),
3217
+ "source_manifest_path": optimization.get("source_manifest_path"),
3218
+ }
3219
+
3220
+
3221
+ def _artifact_action_candidate_records(
3222
+ history_item: Mapping[str, Any],
3223
+ ) -> list[dict[str, Any]]:
3224
+ report = _as_mapping(history_item.get("report"))
3225
+ records: list[dict[str, Any]] = []
3226
+ for child in _as_list(report.get("children") or report.get("jobs")):
3227
+ child_item = _as_mapping(child)
3228
+ if str(child_item.get("command") or "").replace("-", "_") != "action_run":
3229
+ continue
3230
+ action_result = _as_mapping(child_item.get("result"))
3231
+ action_summary = _as_mapping(action_result.get("summary"))
3232
+ action_id = str(
3233
+ action_summary.get("action_id")
3234
+ or _artifact_action_id_from_patch(history_item)
3235
+ or child_item.get("id")
3236
+ or ""
3237
+ )
3238
+ output_count = int(action_summary.get("output_count") or 0)
3239
+ outputs_written_count = int(action_summary.get("outputs_written_count") or 0)
3240
+ completion = _artifact_action_completion_rate(
3241
+ action_summary,
3242
+ output_count=output_count,
3243
+ outputs_written_count=outputs_written_count,
3244
+ )
3245
+ action_kind = str(action_summary.get("action_kind") or "cli")
3246
+ evidence_denominator = 1.0 if action_kind == "download" else 4.0
3247
+ evidence_depth = round(
3248
+ min(outputs_written_count / evidence_denominator, 1.0),
3249
+ 4,
3250
+ )
3251
+ records.append(
3252
+ {
3253
+ "candidate_id": history_item.get("candidate_id"),
3254
+ "action_id": action_id,
3255
+ "action_label": action_summary.get("action_label"),
3256
+ "action_kind": action_kind,
3257
+ "artifact_ref": action_summary.get("artifact_ref"),
3258
+ "source_card_path": action_summary.get("source_card_path"),
3259
+ "score": history_item.get("score"),
3260
+ "action_score": round((0.8 * completion) + (0.2 * evidence_depth), 4),
3261
+ "status": action_result.get("status") or child_item.get("status"),
3262
+ "exit_code": action_result.get("exit_code", child_item.get("exit_code")),
3263
+ "output_count": output_count,
3264
+ "outputs_written_count": outputs_written_count,
3265
+ "output_completion_rate": completion,
3266
+ "evidence_depth": evidence_depth,
3267
+ "outputs_written": list(action_result.get("outputs_written") or []),
3268
+ "outputs": [
3269
+ {
3270
+ "flag": _as_mapping(output).get("flag"),
3271
+ "path": _as_mapping(output).get("path"),
3272
+ "exists": _as_mapping(output).get("exists"),
3273
+ }
3274
+ for output in _as_list(action_result.get("outputs"))
3275
+ if _as_mapping(output)
3276
+ ],
3277
+ "command_args": list(action_result.get("command_args") or []),
3278
+ "patch": copy.deepcopy(dict(history_item.get("patch") or {})),
3279
+ }
3280
+ )
3281
+ return records
3282
+
3283
+
3284
+ def _artifact_action_completion_rate(
3285
+ summary: Mapping[str, Any],
3286
+ *,
3287
+ output_count: int,
3288
+ outputs_written_count: int,
3289
+ ) -> float:
3290
+ if summary.get("output_completion_rate") is not None:
3291
+ return round(float(summary.get("output_completion_rate") or 0.0), 4)
3292
+ if output_count:
3293
+ return round(outputs_written_count / output_count, 4)
3294
+ return 1.0
3295
+
3296
+
3297
+ def _artifact_action_selected_id(
3298
+ optimization: Mapping[str, Any],
3299
+ candidates: Sequence[Mapping[str, Any]],
3300
+ ) -> str | None:
3301
+ best_config = _as_mapping(optimization.get("best_config"))
3302
+ for job in _as_list(best_config.get("jobs")):
3303
+ action_id = _as_mapping(job).get("action_id")
3304
+ if action_id:
3305
+ return str(action_id)
3306
+ if not candidates:
3307
+ return None
3308
+ best = max(candidates, key=lambda record: float(record.get("score") or 0.0))
3309
+ return str(best.get("action_id")) if best.get("action_id") else None
3310
+
3311
+
3312
+ def _artifact_action_id_from_patch(history_item: Mapping[str, Any]) -> str | None:
3313
+ patch = _as_mapping(history_item.get("patch") or history_item.get("candidate_patch"))
3314
+ job = _as_mapping(patch.get("jobs.0"))
3315
+ action_id = job.get("action_id")
3316
+ return str(action_id) if action_id else None
3317
+
3318
+
3319
+ def _artifact_action_selection_reason(selected: Mapping[str, Any]) -> str:
3320
+ action_id = selected.get("action_id") or "selected action"
3321
+ status = selected.get("status") or "unknown"
3322
+ output_count = selected.get("output_count")
3323
+ outputs_written = selected.get("outputs_written_count")
3324
+ completion = selected.get("output_completion_rate")
3325
+ score = selected.get("score")
3326
+ return (
3327
+ f"Selected {action_id} because it finished with status {status}, "
3328
+ f"score {score}, output completion {completion}, and "
3329
+ f"{outputs_written}/{output_count} declared outputs written."
3330
+ )
3331
+
3332
+
3333
+ def _suite_capability_summary(children: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
3334
+ caps: dict[str, set[str]] = {
3335
+ "channels": set(),
3336
+ "child_ids": set(),
3337
+ "commands": set(),
3338
+ "environment_state_keys": set(),
3339
+ "environment_types": set(),
3340
+ "evidence_roles": set(),
3341
+ "evidence_statuses": set(),
3342
+ "frameworks": set(),
3343
+ "metrics": set(),
3344
+ "modalities": set(),
3345
+ "providers": set(),
3346
+ "result_kinds": set(),
3347
+ "search_paths": set(),
3348
+ }
3349
+ for child in children:
3350
+ _add_capability(caps, "child_ids", child.get("id"))
3351
+ _add_capability(caps, "commands", child.get("command"))
3352
+ _add_capability(caps, "result_kinds", child.get("kind"))
3353
+ evidence = _as_mapping(child.get("evidence"))
3354
+ _add_capability(caps, "evidence_roles", evidence.get("role"))
3355
+ _add_capability(caps, "evidence_statuses", evidence.get("status"))
3356
+ result = _as_mapping(child.get("result"))
3357
+ _collect_result_capabilities(result, caps)
3358
+ return {key: sorted(values) for key, values in caps.items()}
3359
+
3360
+
3361
+ def _suite_required_capabilities(suite: Mapping[str, Any]) -> dict[str, list[str]]:
3362
+ raw = (
3363
+ suite.get("required_capabilities")
3364
+ or suite.get("capability_requirements")
3365
+ or suite.get("capabilities_required")
3366
+ or {}
3367
+ )
3368
+ if not isinstance(raw, Mapping):
3369
+ return {}
3370
+ requirements: dict[str, list[str]] = {}
3371
+ for key, values in raw.items():
3372
+ normalized_key = _suite_key(key)
3373
+ if not normalized_key:
3374
+ continue
3375
+ normalized_values = sorted(
3376
+ {
3377
+ _suite_key(value)
3378
+ for value in _as_list(values)
3379
+ if _suite_key(value)
3380
+ }
3381
+ )
3382
+ if normalized_values:
3383
+ requirements[normalized_key] = normalized_values
3384
+ return requirements
3385
+
3386
+
3387
+ def _missing_required_capabilities(
3388
+ required: Mapping[str, Sequence[str]],
3389
+ observed: Mapping[str, Sequence[str]],
3390
+ ) -> dict[str, list[str]]:
3391
+ missing: dict[str, list[str]] = {}
3392
+ for key, required_values in required.items():
3393
+ observed_values = {_suite_key(value) for value in _as_list(observed.get(key))}
3394
+ missing_values = sorted(
3395
+ {
3396
+ _suite_key(value)
3397
+ for value in _as_list(required_values)
3398
+ if _suite_key(value) and _suite_key(value) not in observed_values
3399
+ }
3400
+ )
3401
+ if missing_values:
3402
+ missing[key] = missing_values
3403
+ return missing
3404
+
3405
+
3406
+ def _suite_capability_findings(
3407
+ missing_capabilities: Mapping[str, Sequence[str]],
3408
+ ) -> list[dict[str, Any]]:
3409
+ findings: list[dict[str, Any]] = []
3410
+ for capability, missing_values in sorted(missing_capabilities.items()):
3411
+ values = sorted(_suite_key(value) for value in missing_values if _suite_key(value))
3412
+ if not values:
3413
+ continue
3414
+ findings.append(
3415
+ {
3416
+ "type": "suite_required_capability_missing",
3417
+ "level": "error",
3418
+ "reason": (
3419
+ f"Missing required suite capability `{capability}`: "
3420
+ f"{', '.join(values)}."
3421
+ ),
3422
+ "capability": capability,
3423
+ "missing": values,
3424
+ }
3425
+ )
3426
+ return findings
3427
+
3428
+
3429
+ def _collect_result_capabilities(payload: Mapping[str, Any], caps: dict[str, set[str]]) -> None:
3430
+ for child in _as_list(payload.get("children") or payload.get("jobs")):
3431
+ child_item = _as_mapping(child)
3432
+ if not child_item:
3433
+ continue
3434
+ _add_capability(caps, "child_ids", child_item.get("id"))
3435
+ _add_capability(caps, "commands", child_item.get("command"))
3436
+ _add_capability(caps, "result_kinds", child_item.get("kind"))
3437
+ _collect_result_capabilities(_as_mapping(child_item.get("result")), caps)
3438
+ _collect_summary_capabilities(_as_mapping(payload.get("summary")), caps)
3439
+ optimization = _as_mapping(payload.get("optimization"))
3440
+ best_config = _as_mapping(optimization.get("best_config"))
3441
+ simulation = _as_mapping(best_config.get("simulation"))
3442
+ for environment in _as_list(simulation.get("environments")):
3443
+ env = _as_mapping(environment)
3444
+ _add_capability(caps, "environment_types", env.get("type"))
3445
+ for history in _as_list(optimization.get("history")):
3446
+ item = _as_mapping(history)
3447
+ _add_capabilities(caps, "metrics", _as_mapping(item.get("metrics")).keys())
3448
+ _collect_report_capabilities(_as_mapping(item.get("report")), caps)
3449
+ _collect_report_capabilities(_as_mapping(payload.get("report")), caps)
3450
+ _collect_report_capabilities(_as_mapping(_as_mapping(payload.get("evaluation")).get("report")), caps)
3451
+ _collect_payload_capabilities(payload, caps)
3452
+
3453
+
3454
+ def _collect_report_capabilities(report: Mapping[str, Any], caps: dict[str, set[str]]) -> None:
3455
+ for result in _as_list(report.get("results")):
3456
+ case = _as_mapping(result)
3457
+ metadata = _as_mapping(case.get("metadata"))
3458
+ environment_state = _as_mapping(metadata.get("environment_state"))
3459
+ _add_capabilities(caps, "environment_state_keys", environment_state.keys())
3460
+ for state in environment_state.values():
3461
+ _collect_payload_capabilities(state, caps)
3462
+ _collect_payload_capabilities(_as_mapping(case.get("evaluation")), caps)
3463
+
3464
+
3465
+ def _collect_payload_capabilities(
3466
+ value: Any,
3467
+ caps: dict[str, set[str]],
3468
+ *,
3469
+ depth: int = 0,
3470
+ ) -> None:
3471
+ if depth > 12:
3472
+ return
3473
+ if isinstance(value, Mapping):
3474
+ item = _as_mapping(value)
3475
+ _collect_summary_capabilities(_as_mapping(item.get("summary")), caps)
3476
+ _add_capability(caps, "frameworks", item.get("framework"))
3477
+ _add_capability(caps, "providers", item.get("provider"))
3478
+ _add_capability(caps, "providers", item.get("provider_id"))
3479
+ _add_capability(caps, "providers", item.get("provider_type"))
3480
+ _add_capability(caps, "channels", item.get("channel"))
3481
+ _add_capability(caps, "channels", item.get("modality"))
3482
+ _add_capability(caps, "modalities", item.get("modality"))
3483
+ _add_capabilities(caps, "metrics", _as_mapping(item.get("metrics")).keys())
3484
+ if _suite_key(item.get("type")) in _KNOWN_ENVIRONMENT_TYPES:
3485
+ _add_capability(caps, "environment_types", item.get("type"))
3486
+ for metric in _as_list(item.get("metrics")):
3487
+ metric_item = _as_mapping(metric)
3488
+ _add_capability(caps, "metrics", metric_item.get("name"))
3489
+ for child in item.values():
3490
+ _collect_payload_capabilities(child, caps, depth=depth + 1)
3491
+ elif isinstance(value, list):
3492
+ for child in value:
3493
+ _collect_payload_capabilities(child, caps, depth=depth + 1)
3494
+
3495
+
3496
+ def _collect_summary_capabilities(summary: Mapping[str, Any], caps: dict[str, set[str]]) -> None:
3497
+ if not summary:
3498
+ return
3499
+ _add_capabilities(caps, "search_paths", summary.get("search_paths"))
3500
+ _add_capabilities(caps, "providers", summary.get("observed_providers"))
3501
+ _add_capabilities(caps, "providers", summary.get("required_providers"))
3502
+ _add_capabilities(caps, "channels", summary.get("observed_channels"))
3503
+ _add_capabilities(caps, "channels", summary.get("required_channels"))
3504
+ _add_capabilities(caps, "frameworks", summary.get("trace_frameworks"))
3505
+ _add_capabilities(caps, "frameworks", summary.get("observed_frameworks"))
3506
+ _add_capabilities(caps, "frameworks", summary.get("required_trace_frameworks"))
3507
+ _add_capabilities(caps, "frameworks", summary.get("frameworks"))
3508
+ _add_capabilities(caps, "environment_state_keys", summary.get("environment_state_keys"))
3509
+ evidence_admission = _as_mapping(summary.get("evidence_admission"))
3510
+ _add_capabilities(caps, "evidence_statuses", evidence_admission.get("by_status"))
3511
+ _add_capabilities(caps, "evidence_roles", evidence_admission.get("by_role"))
3512
+ _add_capabilities(caps, "metrics", summary.get("observed_metrics"))
3513
+ _add_capabilities(caps, "metrics", summary.get("required_metrics"))
3514
+ _add_capabilities(caps, "metrics", summary.get("eval_metrics"))
3515
+ _add_capabilities(caps, "metrics", _as_mapping(summary.get("metric_averages")).keys())
3516
+ provider_channels = _as_mapping(summary.get("provider_channels"))
3517
+ _add_capabilities(caps, "providers", provider_channels.keys())
3518
+ for channels in provider_channels.values():
3519
+ _add_capabilities(caps, "channels", channels)
3520
+
3521
+
3522
+ def _add_capabilities(
3523
+ caps: dict[str, set[str]],
3524
+ key: str,
3525
+ values: Any,
3526
+ ) -> None:
3527
+ if isinstance(values, Mapping):
3528
+ values = values.keys()
3529
+ elif values is None:
3530
+ return
3531
+ elif isinstance(values, (str, bytes)):
3532
+ values = [values]
3533
+ else:
3534
+ try:
3535
+ values = list(values)
3536
+ except TypeError:
3537
+ values = [values]
3538
+ for value in values:
3539
+ _add_capability(caps, key, value)
3540
+
3541
+
3542
+ def _add_capability(caps: dict[str, set[str]], key: str, value: Any) -> None:
3543
+ normalized = _suite_key(value)
3544
+ if normalized:
3545
+ caps[key].add(normalized)
3546
+
3547
+
3548
+ def _suite_findings(children: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
3549
+ findings: list[dict[str, Any]] = []
3550
+ for child in children:
3551
+ exit_code = int(child.get("exit_code", 1))
3552
+ if exit_code != 0:
3553
+ findings.append(
3554
+ {
3555
+ "type": "suite_child_failed",
3556
+ "level": "error",
3557
+ "reason": (
3558
+ f"{child.get('command')} {child.get('id')} exited "
3559
+ f"{exit_code}."
3560
+ ),
3561
+ "job": child.get("id"),
3562
+ "command": child.get("command"),
3563
+ "path": child.get("path"),
3564
+ }
3565
+ )
3566
+ for finding in list(child.get("findings") or []):
3567
+ if isinstance(finding, Mapping):
3568
+ copied = copy.deepcopy(dict(finding))
3569
+ copied.setdefault("job", child.get("id"))
3570
+ copied.setdefault("command", child.get("command"))
3571
+ copied.setdefault("path", child.get("path"))
3572
+ findings.append(copied)
3573
+ return findings
3574
+
3575
+
3576
+ def _suite_sarif_findings(result: Mapping[str, Any]) -> list[dict[str, Any]]:
3577
+ findings = []
3578
+ for finding in list(result.get("findings") or []):
3579
+ if isinstance(finding, Mapping):
3580
+ findings.append(copy.deepcopy(dict(finding)))
3581
+ return findings
3582
+
3583
+
3584
+ def _load_child_source(job: Mapping[str, Any], *, base_dir: Path) -> dict[str, Any]:
3585
+ path = _job_path(job, base_dir=base_dir)
3586
+ loaded = _load_json_or_yaml(path)
3587
+ if not isinstance(loaded, Mapping):
3588
+ raise SuiteError(f"suite job source must be an object: {path}")
3589
+ return dict(loaded)
3590
+
3591
+
3592
+ def _suite_jobs(suite: Mapping[str, Any]) -> list[Mapping[str, Any]]:
3593
+ return [dict(job) for job in _as_list(suite.get("jobs"))]
3594
+
3595
+
3596
+ def _job_path(job: Mapping[str, Any], *, base_dir: Path) -> Path:
3597
+ raw = (
3598
+ job.get("path")
3599
+ or job.get("manifest")
3600
+ or job.get("suite")
3601
+ or job.get("file")
3602
+ or job.get("current")
3603
+ or job.get("result")
3604
+ )
3605
+ if not raw:
3606
+ replay_paths = _as_list(job.get("manifests") or job.get("paths"))
3607
+ if replay_paths:
3608
+ raw = replay_paths[0]
3609
+ if not raw:
3610
+ raise SuiteError(f"suite job {job.get('id') or ''} requires path")
3611
+ return _resolve_path(str(raw), base_dir)
3612
+
3613
+
3614
+ def _job_compare_baseline_path(job: Mapping[str, Any], *, base_dir: Path) -> Path:
3615
+ raw = job.get("baseline") or job.get("baseline_path") or job.get("baseline-path")
3616
+ if not raw:
3617
+ raise SuiteError(f"suite compare job {job.get('id') or ''} requires baseline")
3618
+ return _resolve_path(str(raw), base_dir)
3619
+
3620
+
3621
+ def _job_replay_manifest_paths(job: Mapping[str, Any], *, base_dir: Path) -> list[Path]:
3622
+ raw_values = _as_list(
3623
+ job.get("manifests")
3624
+ or job.get("paths")
3625
+ or job.get("path")
3626
+ or job.get("manifest")
3627
+ )
3628
+ paths = [_resolve_path(str(value), base_dir) for value in raw_values if str(value)]
3629
+ if not paths:
3630
+ raise SuiteError(f"suite replay job {job.get('id') or ''} requires manifests")
3631
+ return paths
3632
+
3633
+
3634
+ def _job_optional_path(
3635
+ job: Mapping[str, Any],
3636
+ *,
3637
+ base_dir: Path,
3638
+ keys: Sequence[str],
3639
+ ) -> Optional[Path]:
3640
+ for key in keys:
3641
+ raw = job.get(key)
3642
+ if raw not in (None, ""):
3643
+ return _resolve_path(str(raw), base_dir)
3644
+ return None
3645
+
3646
+
3647
+ def _job_action_id(job: Mapping[str, Any]) -> str:
3648
+ raw = (
3649
+ job.get("action_id")
3650
+ or job.get("action-id")
3651
+ or job.get("action")
3652
+ or job.get("actionId")
3653
+ )
3654
+ if raw in (None, ""):
3655
+ raise SuiteError(f"suite action-run job {job.get('id') or ''} requires action_id")
3656
+ return str(raw)
3657
+
3658
+
3659
+ def _job_action_inputs(job: Mapping[str, Any]) -> dict[str, Any]:
3660
+ raw = job.get("inputs") or job.get("action_inputs") or job.get("action-inputs")
3661
+ if raw in (None, ""):
3662
+ return {}
3663
+ if isinstance(raw, Mapping):
3664
+ return dict(raw)
3665
+ parsed: dict[str, Any] = {}
3666
+ for value in _as_list(raw):
3667
+ text = str(value)
3668
+ if "=" not in text:
3669
+ raise SuiteError(f"suite action-run input must be name=value: {text!r}")
3670
+ key, item = text.split("=", 1)
3671
+ if not key.strip():
3672
+ raise SuiteError(f"suite action-run input has empty name: {text!r}")
3673
+ parsed[key.strip()] = item
3674
+ return parsed
3675
+
3676
+
3677
+ def _job_action_artifact_output(job: Mapping[str, Any]) -> Optional[str]:
3678
+ raw = (
3679
+ job.get("artifact_output")
3680
+ or job.get("artifact-output")
3681
+ or job.get("artifact_output_path")
3682
+ or job.get("artifact-output-path")
3683
+ )
3684
+ if raw in (None, ""):
3685
+ return None
3686
+ return str(raw)
3687
+
3688
+
3689
+ def _job_action_cwd(job: Mapping[str, Any], *, base_dir: Path) -> Path:
3690
+ raw = (
3691
+ job.get("cwd")
3692
+ or job.get("working_dir")
3693
+ or job.get("working-dir")
3694
+ or job.get("workdir")
3695
+ )
3696
+ if raw in (None, ""):
3697
+ return base_dir
3698
+ return _resolve_path(str(raw), base_dir)
3699
+
3700
+
3701
+ def _job_output_paths(job: Mapping[str, Any], base_dir: Path) -> dict[str, list[Path]]:
3702
+ outputs: dict[str, list[Path]] = {
3703
+ "json": [],
3704
+ "junit": [],
3705
+ "sarif": [],
3706
+ "markdown": [],
3707
+ }
3708
+ suite_outputs = dict(job.get("outputs") or {})
3709
+ raw_json = [*_as_list(job.get("output")), *_as_list(suite_outputs.get("json"))]
3710
+ raw_junit = _as_list(suite_outputs.get("junit"))
3711
+ raw_sarif = _as_list(suite_outputs.get("sarif"))
3712
+ raw_markdown = [
3713
+ *_as_list(suite_outputs.get("markdown")),
3714
+ *_as_list(suite_outputs.get("md")),
3715
+ ]
3716
+ for value in raw_json:
3717
+ path = _resolve_path(str(value), base_dir)
3718
+ if path.name.endswith((".junit.xml", ".xml")):
3719
+ outputs["junit"].append(path)
3720
+ elif path.name.endswith((".sarif", ".sarif.json")):
3721
+ outputs["sarif"].append(path)
3722
+ else:
3723
+ outputs["json"].append(path)
3724
+ outputs["junit"].extend(_resolve_path(str(value), base_dir) for value in raw_junit)
3725
+ outputs["sarif"].extend(_resolve_path(str(value), base_dir) for value in raw_sarif)
3726
+ outputs["markdown"].extend(
3727
+ _resolve_path(str(value), base_dir) for value in raw_markdown
3728
+ )
3729
+ return outputs
3730
+
3731
+
3732
+ def _normalize_command(value: Any) -> str:
3733
+ command = str(value or "").strip().lower().replace("-", "_")
3734
+ aliases = {
3735
+ "simulation": "run",
3736
+ "simulate": "run",
3737
+ "evaluation": "eval",
3738
+ "evalartifact": "eval_artifact",
3739
+ "eval_artifacts": "eval_artifact",
3740
+ "eval_report": "eval_artifact",
3741
+ "eval_reports": "eval_artifact",
3742
+ "artifact_eval": "eval_artifact",
3743
+ "artifact_evaluation": "eval_artifact",
3744
+ "evaltask": "eval_task",
3745
+ "eval_tasks": "eval_task",
3746
+ "eval_evidence": "eval_task",
3747
+ "action": "action_run",
3748
+ "actions": "action_run",
3749
+ "actionrun": "action_run",
3750
+ "run_action": "action_run",
3751
+ "task_eval": "eval_task",
3752
+ "task_evaluation": "eval_task",
3753
+ "task_evidence_eval": "eval_task",
3754
+ "red_team": "redteam",
3755
+ "optimization": "optimize",
3756
+ "optimizeeval": "optimize_eval",
3757
+ "optimizesuite": "optimize_suite",
3758
+ "suite_optimization": "optimize_suite",
3759
+ "suite_optimizer": "optimize_suite",
3760
+ "subsuite": "suite",
3761
+ "sub_suite": "suite",
3762
+ "promotion": "promote_to_regression",
3763
+ "regression_promotion": "promote_to_regression",
3764
+ "promote": "promote_to_regression",
3765
+ "minimize": "shrink",
3766
+ "minimize_counterexample": "shrink",
3767
+ }
3768
+ command = aliases.get(command, command)
3769
+ if command not in _CHILD_COMMANDS:
3770
+ allowed = ", ".join(sorted(_CHILD_COMMANDS))
3771
+ raise SuiteError(f"unsupported suite job command: {command}; expected {allowed}")
3772
+ return command
3773
+
3774
+
3775
+ def _normalize_suite_job(job: Mapping[str, Any], index: int) -> dict[str, Any]:
3776
+ item = copy.deepcopy(dict(job))
3777
+ command = _normalize_command(item.get("command") or item.get("type"))
3778
+ path = item.get("path") or item.get("manifest") or item.get("suite")
3779
+ if path in (None, ""):
3780
+ raise ValueError(f"suite job {index} requires a path")
3781
+ item["command"] = command
3782
+ item["path"] = _suite_path_text(path)
3783
+ item["id"] = str(item.get("id") or item.get("name") or f"{command}-{index}")
3784
+ return item
3785
+
3786
+
3787
+ def _suite_path_text(path: str | Path) -> str:
3788
+ return str(path)
3789
+
3790
+
3791
+ def _suite_local_target_text(target: str | Path, *, base_dir: str | Path = ".") -> str:
3792
+ target_text = str(target)
3793
+ module_name, separator, attribute_path = target_text.partition(":")
3794
+ if (
3795
+ separator
3796
+ and attribute_path
3797
+ and (
3798
+ module_name.endswith(".py")
3799
+ or "/" in module_name
3800
+ or "\\" in module_name
3801
+ )
3802
+ ):
3803
+ module_path = Path(module_name).expanduser()
3804
+ if not module_path.is_absolute():
3805
+ module_path = Path(base_dir).expanduser() / module_path
3806
+ return f"{module_path.resolve()}:{attribute_path}"
3807
+ return target_text
3808
+
3809
+
3810
+ def _unique_strings(values: Sequence[Any]) -> list[str]:
3811
+ seen: set[str] = set()
3812
+ result: list[str] = []
3813
+ items = values if isinstance(values, (list, tuple, set)) else _as_list(values)
3814
+ for value in items:
3815
+ text = str(value)
3816
+ if text and text not in seen:
3817
+ seen.add(text)
3818
+ result.append(text)
3819
+ return result
3820
+
3821
+
3822
+ def _optimization_lifecycle_paths(
3823
+ *,
3824
+ optimize_manifest_path: str | Path,
3825
+ workspace_dir: str | Path | None,
3826
+ ) -> dict[str, Path]:
3827
+ manifest_path = Path(optimize_manifest_path).expanduser().resolve()
3828
+ if workspace_dir is None:
3829
+ workspace = (
3830
+ manifest_path.parent.parent
3831
+ if manifest_path.parent.name == "manifests"
3832
+ else manifest_path.parent
3833
+ )
3834
+ else:
3835
+ workspace = Path(workspace_dir).expanduser().resolve()
3836
+ artifacts = workspace / "artifacts"
3837
+ regressions = workspace / "regressions"
3838
+ return {
3839
+ "optimize_manifest": manifest_path,
3840
+ "optimization": artifacts / "optimization.json",
3841
+ "optimization_junit": artifacts / "optimization.junit.xml",
3842
+ "optimization_sarif": artifacts / "optimization.sarif.json",
3843
+ "optimization_markdown": artifacts / "optimization.md",
3844
+ "optimization_report": artifacts / "optimization-report.json",
3845
+ "optimization_report_markdown": artifacts / "optimization-report.md",
3846
+ "promotion": artifacts / "promotion.json",
3847
+ "promotion_report": artifacts / "promotion-report.json",
3848
+ "promotion_report_markdown": artifacts / "promotion-report.md",
3849
+ "regression_manifest": regressions / "optimized-regression.json",
3850
+ "replay": artifacts / "replay.json",
3851
+ "replay_junit": artifacts / "replay.junit.xml",
3852
+ "replay_sarif": artifacts / "replay.sarif.json",
3853
+ "replay_markdown": artifacts / "replay.md",
3854
+ "replay_report": artifacts / "replay-report.json",
3855
+ "replay_report_markdown": artifacts / "replay-report.md",
3856
+ }
3857
+
3858
+
3859
+ def _required_env_cli_args(required_env: Sequence[str]) -> list[str]:
3860
+ args: list[str] = []
3861
+ for key in _unique_strings(required_env):
3862
+ args.extend(["--required-env", key])
3863
+ return args
3864
+
3865
+
3866
+ def _lifecycle_step(
3867
+ step_id: str,
3868
+ label: str,
3869
+ command_args: Sequence[Any],
3870
+ *,
3871
+ outputs: Optional[Mapping[str, Any]] = None,
3872
+ ) -> dict[str, Any]:
3873
+ step = {
3874
+ "id": step_id,
3875
+ "label": label,
3876
+ "kind": "cli",
3877
+ "command": " ".join(shlex.quote(str(arg)) for arg in command_args),
3878
+ "command_args": [str(arg) for arg in command_args],
3879
+ }
3880
+ if outputs:
3881
+ step["outputs"] = {key: str(value) for key, value in outputs.items()}
3882
+ return step
3883
+
3884
+
3885
+ def _write_lifecycle_result_bundle(
3886
+ result: Mapping[str, Any],
3887
+ *,
3888
+ json_path: Path,
3889
+ junit_path: Path,
3890
+ sarif_path: Path,
3891
+ markdown_path: Path,
3892
+ source_path: Path,
3893
+ ) -> list[str]:
3894
+ from fi.alk import simulate
3895
+
3896
+ return [
3897
+ _write_json(json_path, result),
3898
+ _write_text(junit_path, simulate.render_junit(result)),
3899
+ _write_text(sarif_path, simulate.render_sarif(result, manifest_path=source_path)),
3900
+ _write_text(
3901
+ markdown_path,
3902
+ simulate.render_markdown(result, source_path=source_path),
3903
+ ),
3904
+ ]
3905
+
3906
+
3907
+ def _write_lifecycle_report_bundle(
3908
+ report: Mapping[str, Any],
3909
+ *,
3910
+ json_path: Path,
3911
+ markdown_path: Path,
3912
+ source_path: Path,
3913
+ ) -> list[str]:
3914
+ from fi.alk import simulate
3915
+
3916
+ return [
3917
+ _write_json(json_path, report),
3918
+ _write_text(
3919
+ markdown_path,
3920
+ simulate.render_markdown(report, source_path=source_path),
3921
+ ),
3922
+ ]
3923
+
3924
+
3925
+ def _write_json(path: Path, payload: Mapping[str, Any]) -> str:
3926
+ return _write_text(
3927
+ path,
3928
+ json.dumps(payload, indent=2, sort_keys=True, default=str) + "\n",
3929
+ )
3930
+
3931
+
3932
+ def _write_text(path: Path, value: str) -> str:
3933
+ path.parent.mkdir(parents=True, exist_ok=True)
3934
+ path.write_text(value, encoding="utf-8")
3935
+ return str(path)
3936
+
3937
+
3938
+ def _job_name(job: Mapping[str, Any]) -> Optional[str]:
3939
+ value = job.get("name")
3940
+ if value in (None, ""):
3941
+ return None
3942
+ return str(value)
3943
+
3944
+
3945
+ def _job_threshold(
3946
+ job: Mapping[str, Any],
3947
+ suite_options: SuiteRunOptions,
3948
+ ) -> Optional[float]:
3949
+ if job.get("threshold") is not None:
3950
+ return float(job["threshold"])
3951
+ return suite_options.threshold
3952
+
3953
+
3954
+ def _job_max_candidates(
3955
+ job: Mapping[str, Any],
3956
+ suite_options: SuiteRunOptions,
3957
+ ) -> Optional[int]:
3958
+ if job.get("max_candidates") is not None:
3959
+ return int(job["max_candidates"])
3960
+ if job.get("max-candidates") is not None:
3961
+ return int(job["max-candidates"])
3962
+ return suite_options.max_candidates
3963
+
3964
+
3965
+ def _job_dry_run(job: Mapping[str, Any], suite_options: SuiteRunOptions) -> bool:
3966
+ return bool(suite_options.dry_run or job.get("dry_run") or job.get("dry-run"))
3967
+
3968
+
3969
+ def _job_int(
3970
+ job: Mapping[str, Any],
3971
+ *keys: str,
3972
+ default: int,
3973
+ ) -> int:
3974
+ for key in keys:
3975
+ if job.get(key) is not None:
3976
+ return int(job[key])
3977
+ return default
3978
+
3979
+
3980
+ def _job_float(
3981
+ job: Mapping[str, Any],
3982
+ *keys: str,
3983
+ default: float,
3984
+ ) -> float:
3985
+ for key in keys:
3986
+ if job.get(key) is not None:
3987
+ return float(job[key])
3988
+ return default
3989
+
3990
+
3991
+ def _job_optional_float(
3992
+ job: Mapping[str, Any],
3993
+ *keys: str,
3994
+ ) -> Optional[float]:
3995
+ for key in keys:
3996
+ if job.get(key) is not None:
3997
+ return float(job[key])
3998
+ return None
3999
+
4000
+
4001
+ def _merge_options(
4002
+ options: Optional[SuiteRunOptions],
4003
+ *,
4004
+ name: Optional[str] = None,
4005
+ threshold: Optional[float] = None,
4006
+ max_candidates: Optional[int] = None,
4007
+ dry_run: Optional[bool] = None,
4008
+ fail_fast: Optional[bool] = None,
4009
+ require_optimizer_governance: Optional[bool] = None,
4010
+ ) -> SuiteRunOptions:
4011
+ base = options or SuiteRunOptions()
4012
+ return SuiteRunOptions(
4013
+ name=name if name is not None else base.name,
4014
+ threshold=threshold if threshold is not None else base.threshold,
4015
+ max_candidates=(
4016
+ max_candidates if max_candidates is not None else base.max_candidates
4017
+ ),
4018
+ dry_run=dry_run if dry_run is not None else base.dry_run,
4019
+ fail_fast=fail_fast if fail_fast is not None else base.fail_fast,
4020
+ require_optimizer_governance=(
4021
+ require_optimizer_governance
4022
+ if require_optimizer_governance is not None
4023
+ else base.require_optimizer_governance
4024
+ ),
4025
+ )
4026
+
4027
+
4028
+ def _merge_optimization_options(
4029
+ options: Optional[SuiteOptimizationOptions],
4030
+ *,
4031
+ name: Optional[str] = None,
4032
+ threshold: Optional[float] = None,
4033
+ max_candidates: Optional[int] = None,
4034
+ dry_run: Optional[bool] = None,
4035
+ ) -> SuiteOptimizationOptions:
4036
+ base = options or SuiteOptimizationOptions()
4037
+ return SuiteOptimizationOptions(
4038
+ name=name if name is not None else base.name,
4039
+ threshold=threshold if threshold is not None else base.threshold,
4040
+ max_candidates=(
4041
+ max_candidates if max_candidates is not None else base.max_candidates
4042
+ ),
4043
+ dry_run=dry_run if dry_run is not None else base.dry_run,
4044
+ )
4045
+
4046
+
4047
+ def _optimization_cli() -> Any:
4048
+ import importlib
4049
+
4050
+ return importlib.import_module("fi.alk.simulate.cli")
4051
+
4052
+
4053
+ def _load_json_or_yaml(path: Path) -> Any:
4054
+ if path.suffix.lower() in {".yaml", ".yml"}:
4055
+ try:
4056
+ import yaml # type: ignore
4057
+ except Exception as exc: # pragma: no cover - optional dependency clarity
4058
+ raise SuiteError("YAML suite manifests require PyYAML.") from exc
4059
+ with path.open("r", encoding="utf-8") as handle:
4060
+ return yaml.safe_load(handle)
4061
+ with path.open("r", encoding="utf-8") as handle:
4062
+ return json.load(handle)
4063
+
4064
+
4065
+ def _suite_base_dir(suite_path: str | Path) -> Path:
4066
+ path = Path(suite_path).expanduser().resolve()
4067
+ if path.suffix:
4068
+ return path.parent
4069
+ return path
4070
+
4071
+
4072
+ def _resolve_path(value: str, base_dir: Path) -> Path:
4073
+ path = Path(value).expanduser()
4074
+ if path.is_absolute():
4075
+ return path
4076
+ return (base_dir / path).resolve()
4077
+
4078
+
4079
+ def _run_async(awaitable: Any) -> Any:
4080
+ return asyncio.run(awaitable)
4081
+
4082
+
4083
+ def _as_list(value: Any) -> list[Any]:
4084
+ if value is None:
4085
+ return []
4086
+ if isinstance(value, list):
4087
+ return value
4088
+ return [value]
4089
+
4090
+
4091
+ def _as_string_list(value: Any) -> list[str]:
4092
+ return [str(item) for item in _as_list(value) if str(item)]
4093
+
4094
+
4095
+ def _as_mapping(value: Any) -> dict[str, Any]:
4096
+ return dict(value) if isinstance(value, Mapping) else {}
4097
+
4098
+
4099
+ def _suite_key(value: Any) -> str:
4100
+ if isinstance(value, Mapping):
4101
+ return ""
4102
+ if isinstance(value, (list, tuple, set)):
4103
+ return ""
4104
+ return str(value or "").strip().lower().replace("-", "_").replace(" ", "_")
4105
+
4106
+
4107
+ _TRUST_VERDICT_RANK = {
4108
+ "rejected": 0,
4109
+ "conditional": 1,
4110
+ "approved": 2,
4111
+ }
4112
+
4113
+
4114
+ def _optional_bool(value: Any, fallback: Any = None) -> bool | None:
4115
+ candidate = value if value is not None else fallback
4116
+ if candidate is None:
4117
+ return None
4118
+ if isinstance(candidate, bool):
4119
+ return candidate
4120
+ if isinstance(candidate, str):
4121
+ normalized = candidate.strip().lower()
4122
+ if normalized in {"true", "1", "yes"}:
4123
+ return True
4124
+ if normalized in {"false", "0", "no"}:
4125
+ return False
4126
+ return None
4127
+
4128
+
4129
+ _KNOWN_ENVIRONMENT_TYPES = {
4130
+ "adversarial_attack_pack",
4131
+ "agent_control_plane",
4132
+ "agent_integration",
4133
+ "agent_memory_lineage",
4134
+ "agent_trust_boundary",
4135
+ "autonomy_loop",
4136
+ "browser",
4137
+ "domain_package",
4138
+ "framework_capability",
4139
+ "framework_lifecycle",
4140
+ "framework_portability",
4141
+ "framework_probe",
4142
+ "framework_trace",
4143
+ "multimodal_image",
4144
+ "multi_agent_room",
4145
+ "observability_replay",
4146
+ "openenv",
4147
+ "optimizer_trace",
4148
+ "persistent_state_attack",
4149
+ "red_team_campaign",
4150
+ "red_team_readiness",
4151
+ "retrieval_memory",
4152
+ "stateful_tool_world",
4153
+ "streaming_trace",
4154
+ "voice",
4155
+ "workspace_run_manifest",
4156
+ "world_attack_replay",
4157
+ "world_contract",
4158
+ "world_orchestration_replay",
4159
+ }
4160
+
4161
+
4162
+ def _md_cell(value: Any) -> str:
4163
+ return str(value).replace("|", "\\|").replace("\n", " ")
4164
+
4165
+
4166
+ __all__ = [
4167
+ "AGENT_LEARNING_OPTIMIZATION_LIFECYCLE_KIND",
4168
+ "AGENT_LEARNING_SUITE_KIND",
4169
+ "AGENT_LEARNING_SUITE_OPTIMIZATION_KIND",
4170
+ "AGENT_LEARNING_SUITE_TRUST_CERTIFICATE_KIND",
4171
+ "AGENT_LEARNING_SUITE_TRUST_VERIFICATION_KIND",
4172
+ "SuiteError",
4173
+ "SuiteOptimizationOptions",
4174
+ "SuiteRunOptions",
4175
+ "build_framework_adapter_trinity_suite_optimization_manifest",
4176
+ "build_framework_adapter_trinity_suite_manifest",
4177
+ "build_optimization_lifecycle_plan",
4178
+ "build_regression_artifact_suite_manifest",
4179
+ "build_suite_manifest",
4180
+ "build_trinity_suite_manifest",
4181
+ "load_suite",
4182
+ "load_suite_artifact_file",
4183
+ "load_suite_file",
4184
+ "missing_suite_env",
4185
+ "optimize_suite",
4186
+ "optimize_suite_file",
4187
+ "render_junit",
4188
+ "render_markdown",
4189
+ "render_sarif",
4190
+ "required_suite_env",
4191
+ "run_optimization_lifecycle_file",
4192
+ "run_suite",
4193
+ "run_suite_file",
4194
+ "verify_trust_certificate",
4195
+ "verify_trust_certificate_file",
4196
+ "validate_suite_env",
4197
+ "write_framework_adapter_trinity_suite_optimization_workspace",
4198
+ "write_framework_adapter_trinity_suite_workspace",
4199
+ "write_suite_file",
4200
+ ]