agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,903 @@
1
+ from __future__ import annotations
2
+
3
+ import copy
4
+ from typing import Any, Dict, Mapping, Optional, Sequence
5
+ from urllib.parse import urlparse
6
+
7
+ from fi.simulate.environment import (
8
+ AgentMemoryLineageEnvironment,
9
+ FrameworkTraceEnvironment,
10
+ MultiAgentRoomEnvironment,
11
+ RetrievalMemoryEnvironment,
12
+ WorldContractEnvironment,
13
+ WorldOrchestrationReplayEnvironment,
14
+ )
15
+
16
+
17
+ DEFAULT_ORCHESTRATION_PROBE_TOOLS = (
18
+ "apply_world_transition",
19
+ "framework_trace_status",
20
+ "retrieve_documents",
21
+ "read_document",
22
+ "cite_sources",
23
+ "agent_memory_lineage_status",
24
+ "retrieval_memory_status",
25
+ "room_status",
26
+ "request_review",
27
+ "reconcile",
28
+ )
29
+
30
+
31
+ def orchestration_stack_contract(
32
+ *,
33
+ target: str | None = None,
34
+ metadata: Optional[Dict[str, Any]] = None,
35
+ external_sources: Sequence[str] = (),
36
+ environment_types: Sequence[str] = (),
37
+ ) -> dict[str, Any]:
38
+ """Return an import-free local contract for a whole orchestration stack."""
39
+
40
+ target_scheme = urlparse(str(target or "")).scheme.lower()
41
+ external_source_list = _unique_strings(external_sources)
42
+ requires_external = target_scheme in {"http", "https"} or bool(external_source_list)
43
+ return {
44
+ "kind": "agent-learning.orchestration-stack-contract.v1",
45
+ "runtime": "in_process",
46
+ "target": str(target) if target else "",
47
+ "target_scheme": target_scheme,
48
+ "requires_external_service": requires_external,
49
+ "local_executable_fixture": not requires_external,
50
+ "environment_types": _unique_strings(environment_types),
51
+ "external_sources": external_source_list,
52
+ "evidence_requirements": [
53
+ "world_contract",
54
+ "world_transition",
55
+ "framework_trace",
56
+ "retrieval_memory",
57
+ "current_source_citation",
58
+ "agent_memory_lineage",
59
+ "memory_governance",
60
+ "multi_agent_room",
61
+ "critic_review",
62
+ "reconciliation",
63
+ "tool_execution",
64
+ "trace_artifact",
65
+ ],
66
+ "metadata": _plain_mapping(metadata),
67
+ }
68
+
69
+
70
+ def run_orchestration_stack_probe(
71
+ stack: Mapping[str, Any],
72
+ **kwargs: Any,
73
+ ) -> dict[str, Any]:
74
+ """Compatibility alias for the synchronous orchestration stack probe."""
75
+
76
+ return probe_orchestration_stack(stack=stack, **kwargs)
77
+
78
+
79
+ def probe_orchestration_stack(
80
+ *,
81
+ stack: Mapping[str, Any],
82
+ agent: Optional[Mapping[str, Any]] = None,
83
+ target: str | None = None,
84
+ metadata: Optional[Dict[str, Any]] = None,
85
+ allow_external_target: bool = False,
86
+ expected_transition: str = "approve_refund",
87
+ expected_state: Optional[Mapping[str, Any]] = None,
88
+ expected_document_id: str = "doc_refund_2026",
89
+ expected_roles: Sequence[str] = ("planner", "retriever", "critic"),
90
+ expected_review_target: str = "refund",
91
+ expected_reconciliation: str = "approved refund",
92
+ required_tools: Sequence[str] = DEFAULT_ORCHESTRATION_PROBE_TOOLS,
93
+ ) -> dict[str, Any]:
94
+ """Probe local world/framework/retrieval/memory/multi-agent stack evidence."""
95
+
96
+ if target and _is_external_target(target) and not allow_external_target:
97
+ raise ValueError(
98
+ "external targets are disabled for orchestration stack probes; "
99
+ "set allow_external_target=True only when the user explicitly "
100
+ "wants to test that live workload"
101
+ )
102
+ stack_data = _orchestration_stack_data(stack)
103
+ external_sources = _external_sources(stack_data)
104
+ if external_sources and not allow_external_target:
105
+ raise ValueError(
106
+ "external export sources are disabled for orchestration stack probes; "
107
+ "set allow_external_target=True only when the user explicitly "
108
+ "wants to test live exports"
109
+ )
110
+ contract = orchestration_stack_contract(
111
+ target=target,
112
+ metadata=metadata,
113
+ external_sources=external_sources,
114
+ environment_types=[item["type"] for item in stack_data["environments"]],
115
+ )
116
+
117
+ environments = _stack_environments(stack_data["environments"])
118
+ for _environment_type, environment, _data in environments:
119
+ environment.reset()
120
+
121
+ active_agent = agent or _default_orchestration_probe_agent(
122
+ expected_transition=expected_transition,
123
+ expected_document_id=expected_document_id,
124
+ expected_review_target=expected_review_target,
125
+ expected_reconciliation=expected_reconciliation,
126
+ )
127
+ tool_calls = _agent_tool_calls(active_agent)
128
+ handled_tool_calls = 0
129
+ successful_tool_calls = 0
130
+ failed_tool_calls = 0
131
+ observed_tool_names: list[str] = []
132
+ handled_tool_names: list[str] = []
133
+
134
+ for turn_index, tool_call in enumerate(tool_calls, start=1):
135
+ name = str(tool_call.get("name") or "")
136
+ if name:
137
+ observed_tool_names.append(name)
138
+ handled = False
139
+ success = False
140
+ for _environment_type, environment, _data in environments:
141
+ result = environment.handle_tool_call(tool_call, turn_index=turn_index)
142
+ if result is None:
143
+ continue
144
+ handled = True
145
+ success = success or bool(getattr(result, "success", True))
146
+ if handled:
147
+ handled_tool_calls += 1
148
+ handled_tool_names.append(name)
149
+ if success:
150
+ successful_tool_calls += 1
151
+ else:
152
+ failed_tool_calls += 1
153
+
154
+ state = _environment_state(environments)
155
+ summary = _orchestration_probe_summary(
156
+ state,
157
+ stack_data,
158
+ contract=contract,
159
+ tool_calls=tool_calls,
160
+ handled_tool_calls=handled_tool_calls,
161
+ successful_tool_calls=successful_tool_calls,
162
+ failed_tool_calls=failed_tool_calls,
163
+ observed_tool_names=observed_tool_names,
164
+ handled_tool_names=handled_tool_names,
165
+ expected_transition=expected_transition,
166
+ expected_state=expected_state or {"refund.status": "approved"},
167
+ expected_document_id=expected_document_id,
168
+ expected_roles=expected_roles,
169
+ expected_review_target=expected_review_target,
170
+ expected_reconciliation=expected_reconciliation,
171
+ required_tools=required_tools,
172
+ )
173
+ findings = _orchestration_probe_findings(summary, contract=contract)
174
+ summary["finding_count"] = len(findings)
175
+ summary["passed_case_count"] = 1 if not findings else 0
176
+ summary["failed_case_count"] = 0 if not findings else 1
177
+ status = "passed" if not findings else "failed"
178
+ return {
179
+ "kind": "agent-learning.orchestration-stack-probe.v1",
180
+ "status": status,
181
+ "passed": status == "passed",
182
+ "requires_external_service": bool(contract["requires_external_service"]),
183
+ "allow_external_target": bool(allow_external_target),
184
+ "contract": contract,
185
+ "summary": summary,
186
+ "stack": stack_data,
187
+ "environments": copy.deepcopy(stack_data["environments"]),
188
+ "state": state,
189
+ "findings": findings,
190
+ "metadata": {
191
+ "source": "fi.simulate.agent.orchestration.probe_orchestration_stack",
192
+ **_plain_mapping(metadata),
193
+ },
194
+ }
195
+
196
+
197
+ _ORCHESTRATION_ENVIRONMENT_ALIASES: tuple[tuple[tuple[str, ...], str], ...] = (
198
+ (
199
+ ("world_orchestration_replay", "world_replay", "world_orchestration"),
200
+ "world_orchestration_replay",
201
+ ),
202
+ (("world_contract", "world"), "world_contract"),
203
+ (("framework_trace", "framework"), "framework_trace"),
204
+ (("retrieval_memory", "retrieval"), "retrieval_memory"),
205
+ (
206
+ ("agent_memory_lineage", "memory_lineage", "lineage"),
207
+ "agent_memory_lineage",
208
+ ),
209
+ (("multi_agent_room", "room", "multi_agent"), "multi_agent_room"),
210
+ )
211
+
212
+
213
+ def _orchestration_stack_data(stack: Mapping[str, Any]) -> dict[str, Any]:
214
+ source = copy.deepcopy(dict(stack or {}))
215
+ explicit_environments = source.pop("environments", None)
216
+ metadata = _plain_mapping(source.pop("metadata", None))
217
+ name = str(source.pop("name", source.pop("id", "")) or "")
218
+ source.pop("description", None)
219
+ source.pop("target", None)
220
+ source.pop("allow_external_target", None)
221
+ if explicit_environments is not None:
222
+ environments = _environment_list(explicit_environments)
223
+ else:
224
+ environments = []
225
+ for aliases, environment_type in _ORCHESTRATION_ENVIRONMENT_ALIASES:
226
+ data = _pop_first(source, aliases)
227
+ if data is not None:
228
+ environments.append(_typed_environment(environment_type, data))
229
+ if source:
230
+ raise ValueError(
231
+ "orchestration stack has unsupported key(s): "
232
+ f"{', '.join(sorted(source))}"
233
+ )
234
+ if not environments:
235
+ raise ValueError("orchestration stack must define at least one environment")
236
+ return {
237
+ "name": name,
238
+ "metadata": metadata,
239
+ "environments": environments,
240
+ **{
241
+ item["type"]: copy.deepcopy(item.get("data", {}))
242
+ for item in environments
243
+ if item.get("type")
244
+ },
245
+ }
246
+
247
+
248
+ def _environment_list(environments: Any) -> list[dict[str, Any]]:
249
+ if isinstance(environments, Mapping):
250
+ environments = [environments]
251
+ if isinstance(environments, (str, bytes)) or environments is None:
252
+ raise ValueError("environments must be a mapping or sequence of mappings")
253
+ result: list[dict[str, Any]] = []
254
+ for index, raw in enumerate(environments, start=1):
255
+ if not isinstance(raw, Mapping):
256
+ raise ValueError(f"environment {index} must be a mapping")
257
+ item = copy.deepcopy(dict(raw))
258
+ environment_type = _scope_key(item.get("type"))
259
+ if not environment_type:
260
+ raise ValueError(f"environment {index} requires type")
261
+ if item.get("data") is None:
262
+ data = {
263
+ key: value
264
+ for key, value in item.items()
265
+ if key not in {"type", "kind", "name", "description"}
266
+ }
267
+ else:
268
+ data = item["data"]
269
+ result.append(_typed_environment(environment_type, data))
270
+ return result
271
+
272
+
273
+ def _typed_environment(environment_type: str, data: Any) -> dict[str, Any]:
274
+ if not isinstance(data, Mapping):
275
+ raise ValueError(f"{environment_type} candidate data must be a mapping")
276
+ return {"type": _scope_key(environment_type), "data": copy.deepcopy(dict(data))}
277
+
278
+
279
+ def _stack_environments(
280
+ environments: Sequence[Mapping[str, Any]],
281
+ ) -> list[tuple[str, Any, dict[str, Any]]]:
282
+ result: list[tuple[str, Any, dict[str, Any]]] = []
283
+ for item in environments:
284
+ environment_type = _scope_key(item.get("type"))
285
+ data = _plain_mapping(item.get("data"))
286
+ if environment_type == "world_contract":
287
+ result.append((environment_type, WorldContractEnvironment(**data), data))
288
+ elif environment_type == "world_orchestration_replay":
289
+ result.append((environment_type, WorldOrchestrationReplayEnvironment(**data), data))
290
+ elif environment_type == "framework_trace":
291
+ source = dict(data)
292
+ framework = str(source.pop("framework", source.pop("provider", "traceai")))
293
+ result.append(
294
+ (
295
+ environment_type,
296
+ FrameworkTraceEnvironment(framework=framework, **source),
297
+ data,
298
+ )
299
+ )
300
+ elif environment_type == "retrieval_memory":
301
+ source = dict(data)
302
+ documents = source.pop("documents", source.pop("docs", []))
303
+ result.append(
304
+ (
305
+ environment_type,
306
+ RetrievalMemoryEnvironment(
307
+ documents,
308
+ memory=_plain_mapping(source.pop("memory", None)),
309
+ top_k=_as_int(source.pop("top_k", 3)) or 3,
310
+ require_current=bool(source.pop("require_current", True)),
311
+ metadata=_plain_mapping(source.pop("metadata", None)),
312
+ ),
313
+ data,
314
+ )
315
+ )
316
+ elif environment_type == "agent_memory_lineage":
317
+ result.append(
318
+ (
319
+ environment_type,
320
+ AgentMemoryLineageEnvironment(data),
321
+ data,
322
+ )
323
+ )
324
+ elif environment_type == "multi_agent_room":
325
+ source = dict(data)
326
+ participants = (
327
+ source.pop("participants", None)
328
+ or source.pop("agents", None)
329
+ or source.pop("roles", None)
330
+ )
331
+ result.append(
332
+ (
333
+ environment_type,
334
+ MultiAgentRoomEnvironment(
335
+ participants,
336
+ handoff_contracts=source.pop("handoff_contracts", None),
337
+ expected_handoffs=source.pop("expected_handoffs", None),
338
+ expected_reviews=source.pop("expected_reviews", None),
339
+ expected_reconciliation=source.pop("expected_reconciliation", None),
340
+ messages=source.pop("messages", None),
341
+ handoffs=source.pop("handoffs", None),
342
+ reviews=source.pop("reviews", None),
343
+ reconciliations=source.pop("reconciliations", None),
344
+ state=_plain_mapping(source.pop("state", None)),
345
+ allow_unknown_roles=bool(source.pop("allow_unknown_roles", True)),
346
+ extra_trace=_plain_mapping(source.pop("extra_trace", None)),
347
+ ),
348
+ data,
349
+ )
350
+ )
351
+ return result
352
+
353
+
354
+ def _environment_state(environments: Sequence[tuple[str, Any, dict[str, Any]]]) -> dict[str, Any]:
355
+ state: dict[str, Any] = {}
356
+ for environment_type, environment, _data in environments:
357
+ payload_factory = getattr(environment, "_state_payload", None)
358
+ if not callable(payload_factory):
359
+ payload_factory = getattr(environment, "_trace_payload", None)
360
+ if not callable(payload_factory):
361
+ payload_factory = getattr(environment, "_payload", None)
362
+ payload = payload_factory() if callable(payload_factory) else {}
363
+ state[_state_key(environment_type)] = copy.deepcopy(payload)
364
+ return state
365
+
366
+
367
+ def _state_key(environment_type: str) -> str:
368
+ if environment_type == "multi_agent_room":
369
+ return "multi_agent"
370
+ return environment_type
371
+
372
+
373
+ def _orchestration_probe_summary(
374
+ state: Mapping[str, Any],
375
+ stack_data: Mapping[str, Any],
376
+ *,
377
+ contract: Mapping[str, Any],
378
+ tool_calls: Sequence[Mapping[str, Any]],
379
+ handled_tool_calls: int,
380
+ successful_tool_calls: int,
381
+ failed_tool_calls: int,
382
+ observed_tool_names: Sequence[str],
383
+ handled_tool_names: Sequence[str],
384
+ expected_transition: str,
385
+ expected_state: Mapping[str, Any],
386
+ expected_document_id: str,
387
+ expected_roles: Sequence[str],
388
+ expected_review_target: str,
389
+ expected_reconciliation: str,
390
+ required_tools: Sequence[str],
391
+ ) -> dict[str, Any]:
392
+ world = _plain_mapping(state.get("world_contract"))
393
+ framework = _plain_mapping(state.get("framework_trace"))
394
+ retrieval = _plain_mapping(state.get("retrieval_memory"))
395
+ lineage = _plain_mapping(state.get("agent_memory_lineage"))
396
+ room = _plain_mapping(state.get("multi_agent"))
397
+ lineage_summary = _plain_mapping(lineage.get("summary"))
398
+ room_state = _plain_mapping(room.get("state"))
399
+ case_state = _plain_mapping(room_state.get("case"))
400
+
401
+ transition_log = [_plain_mapping(item) for item in _plain_list(world.get("transition_log"))]
402
+ completed_transitions = [
403
+ item for item in transition_log if _scope_key(item.get("status")) == "success"
404
+ ]
405
+ framework_spans = [_plain_mapping(item) for item in _plain_list(framework.get("spans"))]
406
+ framework_events = [_plain_mapping(item) for item in _plain_list(framework.get("events"))]
407
+ framework_signals = set(_unique_strings(framework.get("signals")))
408
+ framework_required = _unique_strings(
409
+ _plain_mapping(stack_data.get("framework_trace")).get("adapter_required_signals")
410
+ )
411
+ documents = [_plain_mapping(item) for item in _plain_list(retrieval.get("documents"))]
412
+ current_doc_ids = {
413
+ str(item.get("id") or "")
414
+ for item in documents
415
+ if item.get("current") is True and str(item.get("id") or "")
416
+ }
417
+ citations = [_plain_mapping(item) for item in _plain_list(retrieval.get("citations"))]
418
+ cited_doc_ids = {
419
+ str(doc_id)
420
+ for citation in citations
421
+ for doc_id in _plain_list(citation.get("doc_ids"))
422
+ if str(doc_id or "")
423
+ }
424
+ required_operations = {"read", "write", "recall"}
425
+ operation_types = set(_plain_list(lineage_summary.get("operation_types")))
426
+ participants = _unique_strings(room.get("participants"))
427
+ reviews = [_plain_mapping(item) for item in _plain_list(room.get("reviews"))]
428
+ reconciliations = [
429
+ _plain_mapping(item) for item in _plain_list(room.get("reconciliations"))
430
+ ]
431
+ required_tool_names = _unique_strings(required_tools)
432
+ observed_tool_name_set = set(_unique_strings(observed_tool_names))
433
+ handled_tool_name_set = set(_unique_strings(handled_tool_names))
434
+ required_roles = set(_unique_strings(expected_roles))
435
+ participant_set = set(participants)
436
+ terminal_status = _scope_key(case_state.get("status") or room_state.get("status"))
437
+ expected_review_present = any(
438
+ _scope_key(item.get("reviewer")) == _scope_key("critic")
439
+ and _scope_key(expected_review_target) in _scope_key(item.get("target"))
440
+ for item in reviews
441
+ )
442
+ expected_reconciliation_present = any(
443
+ _scope_key(expected_reconciliation) in _scope_key(
444
+ item.get("summary") or item.get("decision")
445
+ )
446
+ for item in reconciliations
447
+ )
448
+ return {
449
+ "case_count": 1,
450
+ "passed_case_count": 0,
451
+ "failed_case_count": 1,
452
+ "finding_count": 0,
453
+ "environment_types": _unique_strings(
454
+ [item["type"] for item in stack_data["environments"]]
455
+ ),
456
+ "requires_external_service": bool(contract.get("requires_external_service")),
457
+ "local_executable_fixture": bool(contract.get("local_executable_fixture")),
458
+ "world_present": bool(world),
459
+ "world_transition_count": len(_plain_list(world.get("transitions"))),
460
+ "world_completed_transition_count": len(completed_transitions),
461
+ "expected_transition": str(expected_transition),
462
+ "expected_transition_completed": any(
463
+ _scope_key(item.get("id")) == _scope_key(expected_transition)
464
+ for item in completed_transitions
465
+ ),
466
+ "world_state_match": _state_matches(world.get("state"), expected_state),
467
+ "world_terminal_success": any(
468
+ item.get("pass") is True
469
+ for item in _plain_list(world.get("success_results"))
470
+ ),
471
+ "framework_present": bool(framework),
472
+ "framework": str(framework.get("framework") or ""),
473
+ "framework_span_count": len(framework_spans),
474
+ "framework_event_count": len(framework_events),
475
+ "framework_signal_count": len(framework_signals),
476
+ "framework_required_signal_count": len(framework_required),
477
+ "framework_required_signal_match_count": len(
478
+ set(framework_required) & framework_signals
479
+ ),
480
+ "framework_tool_signal_present": "tool" in framework_signals
481
+ or any(_plain_list(item.get("tool_calls")) for item in framework_spans),
482
+ "retrieval_present": bool(retrieval),
483
+ "retrieval_document_count": len(documents),
484
+ "retrieval_current_document_count": len(current_doc_ids),
485
+ "retrieval_citation_count": len(citations),
486
+ "retrieval_cited_document_count": len(cited_doc_ids),
487
+ "retrieval_citations_current": bool(cited_doc_ids)
488
+ and cited_doc_ids.issubset(current_doc_ids),
489
+ "retrieval_expected_document_id": str(expected_document_id),
490
+ "retrieval_expected_document_cited": str(expected_document_id) in cited_doc_ids,
491
+ "retrieval_freshness_checked_count": sum(
492
+ 1 for citation in citations if citation.get("freshness_checked") is True
493
+ ),
494
+ "memory_present": bool(lineage),
495
+ "memory_store_count": _as_int(lineage_summary.get("store_count")),
496
+ "memory_record_count": _as_int(lineage_summary.get("memory_count")),
497
+ "memory_operation_count": _as_int(lineage_summary.get("operation_count")),
498
+ "memory_audited_operation_count": _as_int(
499
+ lineage_summary.get("audited_operation_count")
500
+ ),
501
+ "memory_required_operations_present": required_operations.issubset(
502
+ operation_types
503
+ ),
504
+ "memory_operation_types": sorted(operation_types),
505
+ "has_source_attribution": bool(lineage_summary.get("has_source_attribution")),
506
+ "has_tenant_isolation": bool(lineage_summary.get("has_tenant_isolation")),
507
+ "has_audit": bool(lineage_summary.get("has_audit")),
508
+ "has_retention_policy": bool(lineage_summary.get("has_retention_policy")),
509
+ "has_deletion_policy": bool(lineage_summary.get("has_deletion_policy")),
510
+ "has_redaction": bool(lineage_summary.get("has_redaction")),
511
+ "has_canaries": bool(lineage_summary.get("has_canaries")),
512
+ "has_observability": bool(lineage_summary.get("has_observability")),
513
+ "has_artifacts": bool(lineage_summary.get("has_artifacts")),
514
+ "policy_violation_count": _as_int(lineage_summary.get("policy_violation_count")),
515
+ "open_poisoning_count": _as_int(lineage_summary.get("open_poisoning_count")),
516
+ "isolation_violation_count": _as_int(
517
+ lineage_summary.get("isolation_violation_count")
518
+ ),
519
+ "retention_violation_count": _as_int(
520
+ lineage_summary.get("retention_violation_count")
521
+ ),
522
+ "blocking_gap_count": _as_int(lineage_summary.get("blocking_gap_count")),
523
+ "room_present": bool(room),
524
+ "participant_count": len(participants),
525
+ "participants": participants,
526
+ "required_roles": sorted(required_roles),
527
+ "role_match": required_roles.issubset(participant_set),
528
+ "allow_unknown_roles": bool(
529
+ _plain_mapping(stack_data.get("multi_agent_room")).get("allow_unknown_roles", True)
530
+ ),
531
+ "review_count": len(reviews),
532
+ "reconciliation_count": len(reconciliations),
533
+ "expected_review_present": expected_review_present,
534
+ "expected_reconciliation_present": expected_reconciliation_present,
535
+ "reconciliation_conflict_count": sum(
536
+ len(_plain_list(item.get("conflicts"))) for item in reconciliations
537
+ ),
538
+ "terminal_room_state": terminal_status in {
539
+ "approved",
540
+ "complete",
541
+ "completed",
542
+ "closed",
543
+ "done",
544
+ "resolved",
545
+ },
546
+ "terminal_status": terminal_status,
547
+ "tool_call_count": len(tool_calls),
548
+ "handled_tool_call_count": int(handled_tool_calls),
549
+ "successful_tool_call_count": int(successful_tool_calls),
550
+ "failed_tool_call_count": int(failed_tool_calls),
551
+ "observed_tool_names": _unique_strings(observed_tool_names),
552
+ "handled_tool_names": _unique_strings(handled_tool_names),
553
+ "required_tool_count": len(required_tool_names),
554
+ "required_tools_present": set(required_tool_names).issubset(
555
+ observed_tool_name_set
556
+ ),
557
+ "required_tools_handled": set(required_tool_names).issubset(
558
+ handled_tool_name_set
559
+ ),
560
+ }
561
+
562
+
563
+ def _orchestration_probe_findings(
564
+ summary: Mapping[str, Any],
565
+ *,
566
+ contract: Mapping[str, Any],
567
+ ) -> list[dict[str, Any]]:
568
+ findings: list[dict[str, Any]] = []
569
+ _append_finding(
570
+ findings,
571
+ "orchestration_probe_local_contract",
572
+ bool(summary.get("local_executable_fixture"))
573
+ and not bool(summary.get("requires_external_service")),
574
+ "orchestration probe target must be local and no-external-service",
575
+ {"contract": dict(contract)},
576
+ )
577
+ _append_finding(
578
+ findings,
579
+ "orchestration_probe_environment_bundle",
580
+ all(
581
+ bool(summary.get(key))
582
+ for key in (
583
+ "world_present",
584
+ "framework_present",
585
+ "retrieval_present",
586
+ "memory_present",
587
+ "room_present",
588
+ )
589
+ ),
590
+ "stack must include world, framework, retrieval, memory lineage, and room evidence",
591
+ summary,
592
+ )
593
+ _append_finding(
594
+ findings,
595
+ "orchestration_probe_world_transition",
596
+ summary.get("expected_transition_completed") is True
597
+ and summary.get("world_state_match") is True
598
+ and summary.get("world_terminal_success") is True,
599
+ "world contract must complete the expected transition and terminal state",
600
+ summary,
601
+ )
602
+ _append_finding(
603
+ findings,
604
+ "orchestration_probe_framework_trace",
605
+ _as_int(summary.get("framework_span_count")) > 0
606
+ and _as_int(summary.get("framework_required_signal_match_count"))
607
+ >= _as_int(summary.get("framework_required_signal_count"))
608
+ and summary.get("framework_tool_signal_present") is True,
609
+ "framework trace must include spans, required signals, and tool evidence",
610
+ summary,
611
+ )
612
+ _append_finding(
613
+ findings,
614
+ "orchestration_probe_retrieval_grounding",
615
+ _as_int(summary.get("retrieval_current_document_count")) > 0
616
+ and _as_int(summary.get("retrieval_citation_count")) > 0
617
+ and summary.get("retrieval_citations_current") is True
618
+ and summary.get("retrieval_expected_document_cited") is True
619
+ and _as_int(summary.get("retrieval_freshness_checked_count"))
620
+ >= _as_int(summary.get("retrieval_citation_count")),
621
+ "retrieval must cite the expected current document with freshness checks",
622
+ summary,
623
+ )
624
+ _append_finding(
625
+ findings,
626
+ "orchestration_probe_memory_lineage_governance",
627
+ _as_int(summary.get("memory_record_count")) > 0
628
+ and summary.get("memory_required_operations_present") is True
629
+ and _as_int(summary.get("memory_audited_operation_count"))
630
+ >= _as_int(summary.get("memory_operation_count"))
631
+ and summary.get("has_source_attribution") is True
632
+ and all(
633
+ summary.get(key) is True
634
+ for key in (
635
+ "has_tenant_isolation",
636
+ "has_audit",
637
+ "has_retention_policy",
638
+ "has_deletion_policy",
639
+ "has_redaction",
640
+ "has_canaries",
641
+ "has_observability",
642
+ "has_artifacts",
643
+ )
644
+ )
645
+ and _as_int(summary.get("policy_violation_count")) == 0
646
+ and _as_int(summary.get("open_poisoning_count")) == 0
647
+ and _as_int(summary.get("isolation_violation_count")) == 0
648
+ and _as_int(summary.get("retention_violation_count")) == 0
649
+ and _as_int(summary.get("blocking_gap_count")) == 0,
650
+ "memory lineage must close source attribution and governance checks",
651
+ summary,
652
+ )
653
+ _append_finding(
654
+ findings,
655
+ "orchestration_probe_multi_agent_coordination",
656
+ summary.get("role_match") is True
657
+ and summary.get("allow_unknown_roles") is False
658
+ and _as_int(summary.get("review_count")) > 0
659
+ and _as_int(summary.get("reconciliation_count")) > 0
660
+ and summary.get("expected_review_present") is True
661
+ and summary.get("expected_reconciliation_present") is True
662
+ and _as_int(summary.get("reconciliation_conflict_count")) == 0
663
+ and summary.get("terminal_room_state") is True,
664
+ "multi-agent room must close roles, review, reconciliation, and terminal state",
665
+ summary,
666
+ )
667
+ _append_finding(
668
+ findings,
669
+ "orchestration_probe_tool_evidence",
670
+ _as_int(summary.get("tool_call_count")) > 0
671
+ and summary.get("required_tools_present") is True
672
+ and summary.get("required_tools_handled") is True
673
+ and _as_int(summary.get("successful_tool_call_count"))
674
+ >= _as_int(summary.get("tool_call_count"))
675
+ and _as_int(summary.get("failed_tool_call_count")) == 0,
676
+ "agent must execute and successfully handle all required orchestration tools",
677
+ summary,
678
+ )
679
+ return findings
680
+
681
+
682
+ def _default_orchestration_probe_agent(
683
+ *,
684
+ expected_transition: str,
685
+ expected_document_id: str,
686
+ expected_review_target: str,
687
+ expected_reconciliation: str,
688
+ ) -> dict[str, Any]:
689
+ return {
690
+ "type": "scripted",
691
+ "responses": [
692
+ {
693
+ "content": "Inspecting world and framework orchestration evidence.",
694
+ "tool_calls": [
695
+ {
696
+ "id": "world_transition",
697
+ "name": "apply_world_transition",
698
+ "arguments": {"id": expected_transition},
699
+ },
700
+ {
701
+ "id": "framework_status",
702
+ "name": "framework_trace_status",
703
+ "arguments": {},
704
+ },
705
+ ],
706
+ },
707
+ {
708
+ "content": "Inspecting retrieval and memory lineage evidence.",
709
+ "tool_calls": [
710
+ {
711
+ "id": "retrieve_current_policy",
712
+ "name": "retrieve_documents",
713
+ "arguments": {"query": "current refund policy"},
714
+ },
715
+ {
716
+ "id": "read_current_policy",
717
+ "name": "read_document",
718
+ "arguments": {"id": expected_document_id},
719
+ },
720
+ {
721
+ "id": "cite_current_policy",
722
+ "name": "cite_sources",
723
+ "arguments": {
724
+ "doc_ids": [expected_document_id],
725
+ "claim": "Current policy supports the orchestration decision.",
726
+ "freshness_checked": True,
727
+ },
728
+ },
729
+ {
730
+ "id": "memory_lineage",
731
+ "name": "agent_memory_lineage_status",
732
+ "arguments": {},
733
+ },
734
+ {
735
+ "id": "retrieval_memory",
736
+ "name": "retrieval_memory_status",
737
+ "arguments": {},
738
+ },
739
+ ],
740
+ },
741
+ {
742
+ "content": "Inspecting review and reconciliation evidence.",
743
+ "tool_calls": [
744
+ {
745
+ "id": "room_status",
746
+ "name": "room_status",
747
+ "arguments": {},
748
+ },
749
+ {
750
+ "id": "critic_review",
751
+ "name": "request_review",
752
+ "arguments": {
753
+ "reviewer": "critic",
754
+ "target": expected_review_target,
755
+ "criteria": ["policy", "memory", "world"],
756
+ },
757
+ },
758
+ {
759
+ "id": "reconcile",
760
+ "name": "reconcile",
761
+ "arguments": {
762
+ "summary": expected_reconciliation,
763
+ "accepted_source": "critic",
764
+ "conflicts": [],
765
+ "participants": ["planner", "retriever", "critic"],
766
+ },
767
+ },
768
+ ],
769
+ },
770
+ ],
771
+ }
772
+
773
+
774
+ def _agent_tool_calls(agent: Optional[Mapping[str, Any]]) -> list[dict[str, Any]]:
775
+ if not agent:
776
+ return []
777
+ calls: list[dict[str, Any]] = []
778
+ for response in _plain_list(_plain_mapping(agent).get("responses")):
779
+ for call in _plain_list(_plain_mapping(response).get("tool_calls")):
780
+ item = _plain_mapping(call)
781
+ if item:
782
+ calls.append(item)
783
+ return calls
784
+
785
+
786
+ def _state_matches(state: Any, expected: Mapping[str, Any]) -> bool:
787
+ actual = _plain_mapping(state)
788
+ if not expected:
789
+ return True
790
+ for key, value in expected.items():
791
+ if "." in str(key):
792
+ observed = _lookup_dotted(actual, str(key))
793
+ if observed != value:
794
+ return False
795
+ elif isinstance(value, Mapping):
796
+ if not _state_matches(_plain_mapping(actual.get(key)), value):
797
+ return False
798
+ elif actual.get(key) != value:
799
+ return False
800
+ return True
801
+
802
+
803
+ def _lookup_dotted(state: Mapping[str, Any], path: str) -> Any:
804
+ current: Any = state
805
+ for part in path.split("."):
806
+ if not isinstance(current, Mapping):
807
+ return None
808
+ current = current.get(part)
809
+ return current
810
+
811
+
812
+ def _external_sources(value: Any) -> list[str]:
813
+ sources: list[str] = []
814
+ if isinstance(value, Mapping):
815
+ for key, item in value.items():
816
+ key_text = _scope_key(key)
817
+ if key_text in {
818
+ "export_source",
819
+ "trace_source",
820
+ "source",
821
+ "source_url",
822
+ "voice_export_source",
823
+ } and _is_external_target(str(item)):
824
+ sources.append(str(item))
825
+ sources.extend(_external_sources(item))
826
+ elif isinstance(value, (list, tuple)):
827
+ for item in value:
828
+ sources.extend(_external_sources(item))
829
+ return _unique_strings(sources)
830
+
831
+
832
+ def _append_finding(
833
+ findings: list[dict[str, Any]],
834
+ check: str,
835
+ passed: bool,
836
+ message: str,
837
+ evidence: Mapping[str, Any],
838
+ ) -> None:
839
+ if passed:
840
+ return
841
+ findings.append(
842
+ {
843
+ "check": check,
844
+ "level": "error",
845
+ "message": message,
846
+ "evidence": dict(evidence),
847
+ }
848
+ )
849
+
850
+
851
+ def _pop_first(source: dict[str, Any], keys: Sequence[str]) -> Any:
852
+ for key in keys:
853
+ if key in source:
854
+ return source.pop(key)
855
+ return None
856
+
857
+
858
+ def _plain_mapping(value: Any) -> dict[str, Any]:
859
+ return dict(value) if isinstance(value, Mapping) else {}
860
+
861
+
862
+ def _plain_list(value: Any) -> list[Any]:
863
+ if value is None:
864
+ return []
865
+ if isinstance(value, list):
866
+ return value
867
+ if isinstance(value, tuple):
868
+ return list(value)
869
+ return [value]
870
+
871
+
872
+ def _scope_key(value: Any) -> str:
873
+ return str(value or "").strip().lower().replace("-", "_").replace(" ", "_")
874
+
875
+
876
+ def _unique_strings(values: Any) -> list[str]:
877
+ seen: set[str] = set()
878
+ result: list[str] = []
879
+ for value in _plain_list(values):
880
+ text = str(value)
881
+ if text and text not in seen:
882
+ seen.add(text)
883
+ result.append(text)
884
+ return result
885
+
886
+
887
+ def _as_int(value: Any) -> int:
888
+ try:
889
+ return int(value)
890
+ except (TypeError, ValueError):
891
+ return 0
892
+
893
+
894
+ def _is_external_target(target: str) -> bool:
895
+ return urlparse(str(target)).scheme.lower() in {"http", "https"}
896
+
897
+
898
+ __all__ = [
899
+ "DEFAULT_ORCHESTRATION_PROBE_TOOLS",
900
+ "orchestration_stack_contract",
901
+ "probe_orchestration_stack",
902
+ "run_orchestration_stack_probe",
903
+ ]