agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,144 @@
1
+ """Standing the environment up in containers, with the harness deciding what that means.
2
+
3
+ The harness has read the agent's repository, so it knows what running that agent's code takes:
4
+ which base image, which install command, which store, which services. Encoding any of that here
5
+ would be guessing on behalf of an agent nobody has seen yet, and would be wrong for the next one.
6
+
7
+ So this provides two things and no opinions:
8
+
9
+ - a place to write files, under the session's own ``env`` directory
10
+ - a way to run container commands from there, and read back what happened
11
+
12
+ Everything else, the Dockerfile, the compose file, the schema, the entrypoint, is written by
13
+ whoever read the repository. What is enforced is only what keeps this safe to run on somebody's
14
+ machine: files stay inside the environment directory, and the only commands that run are container
15
+ commands.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ import shlex
22
+ import shutil
23
+ import subprocess
24
+ from pathlib import Path
25
+
26
+ ENV = "env"
27
+
28
+ # Only these. Not a general shell: a tool that can run anything is a tool with no guardrail, and
29
+ # the whole point of routing through here is that what happens is inspectable and bounded.
30
+ ALLOWED = ("docker", "docker-compose")
31
+
32
+ # Long enough for an image build that downloads a base layer, short enough that a hung build is
33
+ # reported rather than waited on forever.
34
+ PATIENCE = 900
35
+
36
+
37
+ def env_root(destination: Path) -> Path:
38
+ """Where this agent's environment definition lives, beside its world."""
39
+ root = Path(destination) / ENV
40
+ root.mkdir(parents=True, exist_ok=True)
41
+ return root
42
+
43
+
44
+ def inside(destination: Path, path: str) -> Path:
45
+ """The full path for a file the harness wants to write, refused if it escapes.
46
+
47
+ A path arrives as text from a model, so it is resolved and then checked rather than trusted.
48
+ Writing outside the environment directory would mean the harness could touch anything on the
49
+ machine it happens to be running on, which is not a thing to leave to a prompt.
50
+ """
51
+ root = env_root(destination).resolve()
52
+ asked = (root / str(path).lstrip("/")).resolve()
53
+ if not asked.is_relative_to(root):
54
+ raise ValueError(
55
+ f"{path!r} is outside the environment directory. Everything the environment needs "
56
+ "lives under env/, so that building it cannot reach the rest of the machine."
57
+ )
58
+ return asked
59
+
60
+
61
+ def write(destination: Path, path: str, contents: str) -> Path:
62
+ """Put one file into the environment definition."""
63
+ target = inside(destination, path)
64
+ target.parent.mkdir(parents=True, exist_ok=True)
65
+ target.write_text(contents, encoding="utf-8")
66
+ return target
67
+
68
+
69
+ def listing(destination: Path) -> list[str]:
70
+ root = env_root(destination)
71
+ return sorted(
72
+ str(found.relative_to(root)) for found in root.rglob("*") if found.is_file()
73
+ )
74
+
75
+
76
+ def available() -> str:
77
+ """Why containers cannot be used here, or an empty string when they can."""
78
+ if not shutil.which("docker"):
79
+ return "docker is not installed, or not on the path"
80
+ done = subprocess.run(
81
+ ["docker", "info", "--format", "{{.ServerVersion}}"],
82
+ capture_output=True,
83
+ text=True,
84
+ timeout=30,
85
+ )
86
+ if done.returncode != 0:
87
+ return (
88
+ f"docker is installed but not running: {(done.stderr or '').strip()[:200]}"
89
+ )
90
+ return ""
91
+
92
+
93
+ def run(
94
+ destination: Path, command: str, *, patience: int = PATIENCE
95
+ ) -> tuple[int, str]:
96
+ """Run one container command from the environment directory.
97
+
98
+ Returns the exit code and the output, both streams together, because a build failure explains
99
+ itself across the two and reading only one is how the actual cause gets lost.
100
+ """
101
+ try:
102
+ words = shlex.split(command)
103
+ except ValueError as exc:
104
+ return 1, f"could not parse container command: {exc}"
105
+ if not words:
106
+ return 1, "no command given"
107
+ if words[0] not in ALLOWED:
108
+ return 1, (
109
+ f"{words[0]!r} is not something this can run. Only {' and '.join(ALLOWED)} commands, "
110
+ "because a general shell here would be a guardrail with nothing behind it. Everything "
111
+ "the environment needs should be in a file it builds from, not in a command."
112
+ )
113
+ blocked = available()
114
+ if blocked:
115
+ return 1, blocked
116
+ # When the daemon is remote (DOCKER_HOST at a socket proxy), a bind mount
117
+ # names a path on the daemon's host — this container's own filesystem is
118
+ # invisible to it. The mount comes up empty and the failure reads as a
119
+ # missing file three steps later, so it is refused here with the reason.
120
+ if os.environ.get("DOCKER_HOST") and (
121
+ " -v " in f" {command} " or "--volume" in command
122
+ ):
123
+ return 1, (
124
+ "bind mounts cannot work in this deployment: the docker daemon runs "
125
+ "outside this container and does not see these paths. Run the script "
126
+ "inline instead (sh -c '<script>'), or COPY files into an image with "
127
+ "a Dockerfile — build contexts do transfer."
128
+ )
129
+ try:
130
+ done = subprocess.run(
131
+ words,
132
+ cwd=str(env_root(destination)),
133
+ capture_output=True,
134
+ text=True,
135
+ timeout=patience,
136
+ )
137
+ except subprocess.TimeoutExpired:
138
+ return 1, (
139
+ f"gave up after {patience}s. An install that takes this long usually means a "
140
+ "dependency is being fetched that is not going to arrive; check what the last step "
141
+ "was trying to reach."
142
+ )
143
+ output = ((done.stdout or "") + (done.stderr or "")).strip()
144
+ return done.returncode, output
fi/alk/image_loop.py ADDED
@@ -0,0 +1,453 @@
1
+ """Phase 9B units 1-4 — the image / multimodal improvement loop (the 13D
2
+ Practice Loop on ``world.kind = image``).
3
+
4
+ ARCH-9B §2.1/§2.2/§2.3/§2.4 / decisions 9B-D1..9B-D6, 9B-A1/A2/A3/A7/A8.
5
+
6
+ This module invents NO optimizer, NO artifact kind, NO loss machinery, NO world.
7
+ It is the IMAGE analogue of ``voice_loop.py`` — a thin composition layer over
8
+ verbatim engines:
9
+
10
+ * the multi-objective image loss compiles via ``loss.compile_objective`` (the
11
+ Goodhart guard at ``loss.py:106-116`` is reused VERBATIM — "There is no
12
+ override."); the 9B-A2 composition rule (>= 2 terms, >= 1 deterministic
13
+ ground-truth anchor — a judge-only loss is INVALID) is a thin validator on
14
+ top, raising ``image_loss_guard_missing`` (``ImageLossCompositionError``);
15
+ * the whole multimodal-agent config is the search space, assembled by
16
+ ``optimize.build_practice_loop_manifest`` (the same ``base_agent`` +
17
+ ``search_space`` whole-agent contract) with ``world.kind=image`` +
18
+ ``task_mode`` (understanding | generation) on ``WorldSpec.spec``;
19
+ * the image sub-attribution is an additive tag stamped alongside the base
20
+ ``FAILURE_LAYERS`` tag (the existing ``practice/_diagnose.py`` machinery is
21
+ consumed, not rewritten);
22
+ * ``world.kind=image`` enters the world-kind space through the R4 registry hook
23
+ (``extensions.register_extension``) — never by widening the frozen
24
+ ``SIMULATION_WORLD_KINDS`` tuple. ``image`` is "typed -> executable".
25
+
26
+ The canon constants below are this module's home; ``trinity.py`` carries literal
27
+ mirrors that the milestone test cross-pins (the GUNA_AXES cross-pin pattern —
28
+ trinity never imports this module so the gate runs even if this is broken). The
29
+ pure-numpy perturbation operators live in the companion ``image_perturb.py``
30
+ (9B-A1b — substrate, not loop).
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ from typing import Any, Mapping, Optional, Sequence
36
+
37
+ # --- canon (ARCH-9B §2.1 image-loss term refs + §2.3 sub-attribution) -------
38
+ # The deterministic-anchored UNDERSTANDING-mode menu (the 6-tuple analogue of
39
+ # V1_VOICE_LOSS_TERM_REFS, the 9-tuple).
40
+ V1_IMAGE_LOSS_TERM_REFS = (
41
+ "task_success",
42
+ "ocr_accuracy",
43
+ "chart_accuracy",
44
+ "artifact_grounding",
45
+ "instruction_adherence",
46
+ "tool_argument_correctness",
47
+ )
48
+ # The mandatory ground-truth quality anchors — an image loss MUST carry >= 1 of
49
+ # these (9B-A2 / 9B-D3). The analogue of V1_VOICE_LOSS_NON_TIMING_QUALITY_TERMS.
50
+ # ``element_presence`` joins this admissible set under task_mode=generation.
51
+ V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS = (
52
+ "task_success",
53
+ "ocr_accuracy",
54
+ "chart_accuracy",
55
+ "artifact_grounding",
56
+ )
57
+ # The bounded/guarded judge contributors (the analogue of V1_VOICE_LOSS_TIMING_
58
+ # TERMS, the hackable-alone set). A judge-only loss (terms subset of this set) is
59
+ # structurally rejected (9B-D3). generation_alignment / generation_quality join
60
+ # this set under task_mode=generation.
61
+ V1_IMAGE_LOSS_JUDGE_TERMS = ("instruction_adherence",)
62
+
63
+ # Generation-mode terms (ARCH-9B §2.4 / 9B-A7/A8), admitted ONLY under
64
+ # task_mode=generation.
65
+ V1_IMAGE_GENERATION_ANCHOR_TERMS = ("element_presence",) # deterministic floor (9B-A8)
66
+ V1_IMAGE_GENERATION_JUDGE_TERMS = ("generation_alignment", "generation_quality")
67
+
68
+ # The four-token image sub-attribution closed set (9B §2.3), stamped alongside
69
+ # the base FAILURE_LAYERS tag.
70
+ V1_IMAGE_FAILURE_SUBLAYERS = ("preprocessing", "perception", "reasoning", "tool_grounding")
71
+
72
+ # A MARKER field on artifact metadata — NOT a new evidence class (R5/A18; the
73
+ # frozen EVIDENCE_CLASSES 4-tuple is unchanged). The analogue of
74
+ # V1_VOICE_FIDELITY_TIERS. (ARCH-9B §2.6)
75
+ V1_IMAGE_FIDELITY_TIERS = ("deterministic_fixture", "keyed_live_model")
76
+
77
+ # The typed ``kind`` discriminators a perception-bypass guard row may carry,
78
+ # beyond the base sentinel/canary rows the loss guard already allows (ARCH-9B
79
+ # §2.2).
80
+ V1_IMAGE_PERCEPTION_GUARD_KINDS = ("perception_bypass", "perceptual_counterfactual")
81
+
82
+ # The task_mode switch on WorldSpec.spec (ARCH-9B §2.3 / 9B-D2). ONE world kind,
83
+ # two loss profiles.
84
+ V1_IMAGE_TASK_MODES = ("understanding", "generation")
85
+
86
+ # The registered world-kind token + the namespaced extension name (R4 hook).
87
+ IMAGE_WORLD_KIND = "image"
88
+ IMAGE_EXTENSION_NAME = "agentlearning.image"
89
+
90
+ # The R4 rung -> evidence-class ladder (ARCH-9B §2.6). The deterministic core is
91
+ # local_gate/captured_fixture; live_lane is added ONLY on the keyed lane record
92
+ # (unit 7), never the day-one deterministic record.
93
+ _IMAGE_RUNG_LADDER = {
94
+ "rung1": ["local_gate"],
95
+ "perturbed": ["live_stressed", "captured_fixture"],
96
+ "keyed_vlm": ["live_lane"],
97
+ }
98
+
99
+
100
+ class ImageLossCompositionError(ValueError):
101
+ """Raised when an image objective violates the 9B-A2 composition rule (the
102
+ ``image_loss_guard_missing`` finding — an image specialization of
103
+ ``objective_guards_missing``). A ``ValueError`` subclass so callers can
104
+ ``except ValueError`` exactly as for ``VoiceLossCompositionError``."""
105
+
106
+
107
+ def _term_refs(objective: Mapping[str, Any]) -> list[str]:
108
+ """The objective's eval refs (read from ``evals`` — the loss.py schema)."""
109
+ return [
110
+ str(term.get("eval"))
111
+ for term in (objective.get("evals") or [])
112
+ if isinstance(term, Mapping) and term.get("eval")
113
+ ]
114
+
115
+
116
+ def _admissible_anchor_terms(task_mode: str) -> tuple[str, ...]:
117
+ if task_mode == "generation":
118
+ return V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS + V1_IMAGE_GENERATION_ANCHOR_TERMS
119
+ return V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS
120
+
121
+
122
+ def _admissible_term_refs(task_mode: str) -> tuple[str, ...]:
123
+ if task_mode == "generation":
124
+ return (
125
+ V1_IMAGE_LOSS_TERM_REFS
126
+ + V1_IMAGE_GENERATION_ANCHOR_TERMS
127
+ + V1_IMAGE_GENERATION_JUDGE_TERMS
128
+ )
129
+ return V1_IMAGE_LOSS_TERM_REFS
130
+
131
+
132
+ def compile_image_objective(
133
+ payload: Mapping[str, Any], *, task_mode: str = "understanding"
134
+ ) -> dict:
135
+ """Compile a multi-objective image loss with a perception-bypass Goodhart
136
+ guard (ARCH-9B §2.2 / 9B-A2 / 9B-D3). The image analogue of
137
+ ``compile_voice_objective`` (voice_loop.py:70). Enforces, ON TOP of the
138
+ verbatim ``loss.compile_objective`` Goodhart guard:
139
+
140
+ (a) >= 2 terms (a single-term image objective is reward-hackable);
141
+ (b) >= 1 deterministic ground-truth anchor — a judge-only loss is INVALID
142
+ (9B-D3). ``task_mode`` selects the admissible anchor set:
143
+ understanding -> V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS; generation
144
+ adds ``element_presence`` (the deterministic floor, 9B-A8);
145
+ (c) unknown-ref rejection (every term must be a member of the mode's menu);
146
+ (d) when sentinel/canary rows carry a perception ``kind`` discriminator it
147
+ must be in V1_IMAGE_PERCEPTION_GUARD_KINDS (the closed set).
148
+
149
+ Then delegates to ``loss.compile_objective`` VERBATIM — which unconditionally
150
+ enforces the populated guard block (sentinel_rows / canary_evals,
151
+ min_guard_count >= 1, "There is no override.")."""
152
+
153
+ from . import loss as _loss # downward facade import (legal; voice_loop.py idiom)
154
+
155
+ if task_mode not in V1_IMAGE_TASK_MODES:
156
+ raise ImageLossCompositionError(
157
+ f"image_loss_guard_missing: task_mode {task_mode!r} not in "
158
+ f"{V1_IMAGE_TASK_MODES}"
159
+ )
160
+
161
+ refs = _term_refs(payload)
162
+
163
+ # rule (a): >= 2 terms.
164
+ if len(refs) < 2:
165
+ raise ImageLossCompositionError(
166
+ "image_loss_guard_missing: an image objective is reward-hackable as a "
167
+ "single term; it MUST be multi-objective (>= 2 terms). "
168
+ f"got {refs}"
169
+ )
170
+
171
+ # rule (b): >= 1 deterministic ground-truth anchor (judge-only REJECTED).
172
+ anchors = _admissible_anchor_terms(task_mode)
173
+ if not any(ref in anchors for ref in refs):
174
+ raise ImageLossCompositionError(
175
+ "image_loss_guard_missing: an image loss MUST carry >= 1 deterministic "
176
+ f"ground-truth anchor {anchors}; a judge-only loss is INVALID by "
177
+ f"contract (9B-D3). got {refs}"
178
+ )
179
+
180
+ # rule (c): unknown-ref rejection.
181
+ allowed = _admissible_term_refs(task_mode)
182
+ for ref in refs:
183
+ if ref not in allowed:
184
+ raise ImageLossCompositionError(
185
+ f"image_loss_guard_missing: unknown image loss term {ref!r}; "
186
+ f"expected members of {allowed} (task_mode={task_mode})"
187
+ )
188
+
189
+ # rule (d): the perception-bypass guard rows ride the existing
190
+ # sentinel_rows/canary_evals with a typed ``kind`` discriminator (no new
191
+ # ObjectiveSpec field, ARCH-9B §2.2). When present it must be in the closed
192
+ # set (plus any untyped/base rows the loss guard already allows).
193
+ guards = payload.get("guards") or {}
194
+ for bucket in ("sentinel_rows", "canary_evals"):
195
+ for row in guards.get(bucket) or []:
196
+ if isinstance(row, Mapping):
197
+ kind = row.get("kind")
198
+ if kind is not None and kind not in V1_IMAGE_PERCEPTION_GUARD_KINDS:
199
+ raise ImageLossCompositionError(
200
+ f"image_loss_guard_missing: guard row kind {kind!r} not in "
201
+ f"{V1_IMAGE_PERCEPTION_GUARD_KINDS}"
202
+ )
203
+
204
+ # the verbatim Goodhart guard (loss.py:106-116) — "There is no override."
205
+ return _loss.compile_objective(payload)
206
+
207
+
208
+ def attribute_image_sublayer(
209
+ *,
210
+ failure_layer: str,
211
+ deficit: Mapping[str, Any] | None = None,
212
+ signal: str | None = None,
213
+ ) -> str:
214
+ """Map a weak image cell to a ``V1_IMAGE_FAILURE_SUBLAYERS`` token, stamped
215
+ ALONGSIDE the base ``FAILURE_LAYERS`` tag (a weak cell carries both, e.g.
216
+ ``{failure_layer:"agent_behavior", image_sublayer:"perception"}``). The base
217
+ attribution rides the existing ``practice/_diagnose.py`` machinery; this is
218
+ the thin sublayer helper (the image analogue of ``attribute_voice_sublayer``,
219
+ voice_loop.py:109).
220
+
221
+ Routing (ARCH-9B §2.3 table, grounded in DISCO's parse->reason split
222
+ 2603.23511 + AgentVista visual-misidentification=perception 2602.23166):
223
+ * OCR/parse weak; low-res/compression hurts -> ``preprocessing``
224
+ (Fix-Before-Search: try a resolution/crop change before blaming the LLM);
225
+ * visual misidentification; perception-required cell weak -> ``perception``
226
+ (the dominant AgentVista failure);
227
+ * grounded-but-wrong-conclusion -> ``reasoning`` (parse ok, reasoning fails);
228
+ * tool-argument extracted wrong from the image -> ``tool_grounding``."""
229
+
230
+ sig = str(signal or (deficit or {}).get("signal") or "").lower()
231
+ if any(
232
+ k in sig
233
+ for k in ("ocr", "parse", "low_res", "low-res", "resolution", "compression",
234
+ "compress", "blur", "preprocess")
235
+ ):
236
+ return "preprocessing"
237
+ if any(
238
+ k in sig
239
+ for k in ("tool_argument", "tool-argument", "tool_grounding", "tool argument",
240
+ "argument", "extracted")
241
+ ):
242
+ return "tool_grounding"
243
+ if any(
244
+ k in sig
245
+ for k in ("misidentif", "perception", "visual", "occlusion", "occluded",
246
+ "perceive", "see ")
247
+ ):
248
+ return "perception"
249
+ if any(
250
+ k in sig
251
+ for k in ("reason", "conclusion", "grounded-but-wrong", "wrong_conclusion",
252
+ "inference")
253
+ ):
254
+ return "reasoning"
255
+ # default: infra-implicated cells land on preprocessing (the cheapest fix
256
+ # before blaming the model); otherwise the reasoning/policy layer.
257
+ if failure_layer in ("lane_infra", "framework_runtime", "provider"):
258
+ return "preprocessing"
259
+ return "reasoning"
260
+
261
+
262
+ def _ensure_image_world_registered() -> None:
263
+ """Register ``world.kind=image`` via the R4 hook (ARCH-9B §2.1 / 9B-D2).
264
+ Idempotent. Pushes DOWN into ``contract.register_world_kind`` so
265
+ ``resolved_world_kinds()`` contains ``image`` WITHOUT touching the frozen
266
+ ``SIMULATION_WORLD_KINDS`` tuple (contract.py:55)."""
267
+
268
+ from fi.simulate.simulation import contract as _contract
269
+
270
+ # idempotent: register_extension raises on a name collision, and the world
271
+ # kind only needs to land once per process.
272
+ if IMAGE_WORLD_KIND in _contract.resolved_world_kinds():
273
+ return
274
+
275
+ from . import extensions as _ext
276
+
277
+ _ext.register_extension(
278
+ "environment",
279
+ {
280
+ "name": IMAGE_EXTENSION_NAME, # vendor.name shape (_validate_record)
281
+ "kind_token": IMAGE_WORLD_KIND, # the registered world.kind token
282
+ "spec_validator": _validate_image_world_spec, # R4 mandate
283
+ "rung_ladder": _IMAGE_RUNG_LADDER, # R4 mandate
284
+ # the deterministic core; live_lane is added ONLY on the keyed lane
285
+ # record (unit 7), never here.
286
+ "evidence_class_capability": ["local_gate", "captured_fixture"],
287
+ # gated_contexts_runnable stays False until rung1_fixture_green
288
+ # (extensions.py admission); 9B never silently claims executable.
289
+ },
290
+ )
291
+
292
+
293
+ def _validate_image_world_spec(spec: Mapping[str, Any]) -> None:
294
+ """The R4 ``spec_validator`` for the image world: validate the ``task_mode``
295
+ switch (understanding | generation) on ``WorldSpec.spec``. Raises ValueError
296
+ on an unknown mode (the closed-set guard)."""
297
+
298
+ task_mode = str((spec or {}).get("task_mode", "understanding"))
299
+ if task_mode not in V1_IMAGE_TASK_MODES:
300
+ raise ValueError(
301
+ f"image world.spec.task_mode {task_mode!r} not in {V1_IMAGE_TASK_MODES}"
302
+ )
303
+
304
+
305
+ def build_image_practice_loop_manifest(
306
+ *,
307
+ name: str,
308
+ base_agent: Mapping[str, Any],
309
+ search_space: Mapping[str, Sequence[Any]],
310
+ objective: Mapping[str, Any],
311
+ eval_budget: int,
312
+ seed: int,
313
+ task_mode: str = "understanding",
314
+ scenario_inline: Optional[Mapping[str, Any]] = None,
315
+ max_rounds: int = 8,
316
+ ) -> dict[str, Any]:
317
+ """Assemble the image improvement-loop manifest: the 13D Practice Loop on
318
+ ``world.kind=image`` + ``task_mode`` with the multi-objective guarded image
319
+ loss + the whole multimodal-agent search space (9B-D5). Delegates to
320
+ ``optimize.build_practice_loop_manifest`` so its validators hold VERBATIM
321
+ (9B-A3). The objective is compiled by ``compile_image_objective`` (the 9B-A2
322
+ rule) before it rides the simulation.
323
+
324
+ Byte-parallel to ``build_voice_practice_loop_manifest`` except: (a) the
325
+ ``_ensure_image_world_registered()`` call (voice's kind is built-in, image's
326
+ is registered through the R4 hook); (b) ``world["kind"]="image"`` instead of
327
+ ``"voice_telephony"``; (c) the ``spec["task_mode"]`` write (voice has no mode
328
+ switch)."""
329
+
330
+ from . import optimize as _optimize # downward facade import (legal)
331
+
332
+ if task_mode not in V1_IMAGE_TASK_MODES:
333
+ raise ImageLossCompositionError(
334
+ f"image_loss_guard_missing: task_mode {task_mode!r} not in "
335
+ f"{V1_IMAGE_TASK_MODES}"
336
+ )
337
+
338
+ _ensure_image_world_registered() # step 1 (§2.1)
339
+ compiled = compile_image_objective(objective, task_mode=task_mode) # step 2 (unit 2)
340
+ inline = dict(scenario_inline or {})
341
+ inline.setdefault("version", "agent-learning.simulation.v1")
342
+ inline["objective"] = compiled
343
+ world = dict(inline.get("world") or {})
344
+ world["kind"] = IMAGE_WORLD_KIND # step 3 — the registered kind
345
+ spec = dict(world.get("spec") or {})
346
+ spec["task_mode"] = task_mode # the task_mode switch on WorldSpec.spec
347
+ world["spec"] = spec
348
+ inline["world"] = world
349
+
350
+ return _optimize.build_practice_loop_manifest( # step 4 — VERBATIM delegate
351
+ name=name,
352
+ simulation={"version": inline["version"], "inline": inline},
353
+ base_agent=base_agent,
354
+ search_space=search_space,
355
+ eval_budget=eval_budget,
356
+ seed=seed,
357
+ max_rounds=max_rounds,
358
+ )
359
+
360
+
361
+ # === Unit 7 — the keyed real-VLM lane (opt-in, NEVER a gate prerequisite) ===
362
+ # ARCH-9B §2.4 / §2.6 / 9B-D1/D6. The judge-anchored terms, the full generation
363
+ # profile, and the one real-multimodal-agent live-proof are owner-keyed, opt-in,
364
+ # never a release gate. The deterministic core stays local_gate/captured_fixture;
365
+ # the keyed lane is the ONLY honest place for live_lane (a real keyed model ran).
366
+
367
+ KEYED_IMAGE_EXTENSION_NAME = "agentlearning.image.keyed"
368
+ _KEYED_IMAGE_RUNG_LADDER = {
369
+ "rung1": ["local_gate"],
370
+ "perturbed": ["live_stressed", "captured_fixture"],
371
+ "keyed_vlm": ["live_lane"],
372
+ }
373
+
374
+ # the env keys that gate the keyed lane (checked, never required by any gate).
375
+ IMAGE_JUDGE_KEY_ENVS = ("AGENT_LEARNING_IMAGE_JUDGE_KEY", "OPENAI_API_KEY")
376
+
377
+
378
+ class ImageKeyedLaneUnavailable(RuntimeError):
379
+ """Raised by the keyed lane when no judge/VLM key is present — the loud
380
+ refusal (the ``image_judge_key_unavailable`` finding). The deterministic core
381
+ NEVER raises this; only the opt-in keyed path does."""
382
+
383
+
384
+ def image_judge_key_present() -> bool:
385
+ """True iff a judge/VLM key is configured for the keyed lane."""
386
+ import os
387
+
388
+ return any(os.environ.get(env) for env in IMAGE_JUDGE_KEY_ENVS)
389
+
390
+
391
+ def register_keyed_image_lane() -> None:
392
+ """Register the SEPARATE keyed-lane extension record that adds ``live_lane``
393
+ to ``evidence_class_capability`` (ARCH-9B §2.6, unit 7). Idempotent. This is
394
+ the ONLY record that may carry ``live_lane`` — the deterministic-core record
395
+ (``_ensure_image_world_registered``) stays ``("local_gate","captured_fixture")``.
396
+ NEVER called by the gate; opt-in only."""
397
+ from . import extensions as _ext
398
+
399
+ if _ext.resolve("environment", KEYED_IMAGE_EXTENSION_NAME) is not None:
400
+ return
401
+ _ext.register_extension(
402
+ "environment",
403
+ {
404
+ "name": KEYED_IMAGE_EXTENSION_NAME,
405
+ # the keyed lane reuses the SAME world.kind token only when the base
406
+ # record is absent; here it declares the keyed capability without a
407
+ # second kind_token (the world kind is already registered).
408
+ "evidence_class_capability": ["local_gate", "captured_fixture", "live_lane"],
409
+ },
410
+ )
411
+
412
+
413
+ def run_keyed_image_live_proof(
414
+ *,
415
+ base_agent: Mapping[str, Any],
416
+ search_space: Mapping[str, Sequence[Any]],
417
+ objective: Mapping[str, Any],
418
+ eval_budget: int,
419
+ seed: int,
420
+ task_mode: str = "generation",
421
+ name: str = "image-keyed-live-proof",
422
+ ) -> dict[str, Any]:
423
+ """The one owner-keyed live-proof entry (WORKFLOW Step 5 real-keys ground
424
+ rule). Refuses LOUDLY without a key (``ImageKeyedLaneUnavailable`` ->
425
+ ``image_judge_key_unavailable``) — never a fake number, never a release
426
+ prerequisite. With a key, it builds the generation-profile manifest and marks
427
+ the run ``live_lane`` / ``fidelity_tier=keyed_live_model``.
428
+
429
+ The keyed run itself (calling the judge/VLM) is left to the caller's runtime;
430
+ this returns the keyed manifest + the honest evidence-class stamp so an
431
+ owner can execute it once with real keys."""
432
+ if not image_judge_key_present():
433
+ raise ImageKeyedLaneUnavailable(
434
+ "image_judge_key_unavailable: the keyed real-VLM lane requires a "
435
+ f"judge/VLM key (one of {IMAGE_JUDGE_KEY_ENVS}); withheld -- never a "
436
+ "fake number, never a release prerequisite"
437
+ )
438
+ register_keyed_image_lane()
439
+ manifest = build_image_practice_loop_manifest(
440
+ name=name,
441
+ base_agent=base_agent,
442
+ search_space=search_space,
443
+ objective=objective,
444
+ eval_budget=eval_budget,
445
+ seed=seed,
446
+ task_mode=task_mode,
447
+ )
448
+ return {
449
+ "manifest": manifest,
450
+ "evidence_class": "live_lane", # the ONLY honest live_lane
451
+ "fidelity_tier": "keyed_live_model",
452
+ "task_mode": task_mode,
453
+ }