agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/cua_loop.py ADDED
@@ -0,0 +1,562 @@
1
+ """Phase 9C units 1-4 — the CUA / browser / computer-use improvement loop (the
2
+ 13D Practice Loop on ``world.kind = browser`` / ``computer_use`` with a
3
+ ``cua_surface = browser | desktop`` sub-kind switch).
4
+
5
+ ARCH-9C §2.1/§2.2/§2.3/§2.4 / decisions 9C-D1..9C-D6, 9C-A1/A1b/A1c/A2/A3/A7/A7b/A8.
6
+
7
+ This module invents NO optimizer, NO artifact kind, NO loss machinery, NO world,
8
+ NO perturbation module. It is the CUA analogue of ``image_loop.py`` /
9
+ ``voice_loop.py`` — a thin composition layer over verbatim engines:
10
+
11
+ * the multi-objective CUA loss compiles via ``loss.compile_objective`` (the
12
+ Goodhart guard at ``loss.py:106-116`` is reused VERBATIM — "There is no
13
+ override."); the 9C-A2 composition rule (>= 2 terms, >= 1 deterministic
14
+ post-state ground-truth anchor — a judge-only loss is INVALID) is a thin
15
+ validator on top, raising ``cua_loss_guard_missing`` (``CuaLossCompositionError``);
16
+ * the whole CUA-agent config is the search space, assembled by
17
+ ``optimize.build_practice_loop_manifest`` (the same ``base_agent`` +
18
+ ``search_space`` whole-agent contract) with ``world.kind=browser`` /
19
+ ``computer_use`` + ``cua_surface`` (browser | desktop) on ``WorldSpec.spec``;
20
+ * the CUA sub-attribution is an additive tag stamped alongside the base
21
+ ``FAILURE_LAYERS`` tag (the existing ``practice/_diagnose.py`` machinery is
22
+ consumed, not rewritten);
23
+ * ``browser`` / ``computer_use`` enter EXECUTABLE-LOOP status through the R4
24
+ registry hook (``extensions.register_extension``) — never by widening the
25
+ frozen ``SIMULATION_WORLD_KINDS`` tuple. They are already FROZEN typed-only
26
+ members (the §0/9C-A1b nuance vs 9B's ``image``): so the registration gates on
27
+ the EXECUTABLE-LOOP RECORD presence in ``_EXTRA_WORLD_KINDS``, NOT on the
28
+ verbatim image idempotence guard (which would short-circuit immediately).
29
+
30
+ The CUA perturbation operators (selector-drift / layout-shift / stale-screenshot
31
+ / injected-DOM) are ALREADY in the kit's ``BrowserEnvironment`` mutation pack
32
+ (9C-A1c / 9C-D4) — there is NO ``cua_perturb.py`` and NO ``apply_cua_perturbations``
33
+ function (the sharpest contrast with 9B's ``image_perturb.py``);
34
+ ``V1_CUA_PERTURBATION_OPERATORS`` is a NAMING MIRROR ONLY.
35
+
36
+ The canon constants below are this module's home; ``trinity.py`` carries literal
37
+ mirrors that the milestone test cross-pins (the GUNA_AXES cross-pin pattern —
38
+ trinity never imports this module so the gate runs even if this is broken).
39
+ """
40
+
41
+ from __future__ import annotations
42
+
43
+ from typing import Any, Mapping, Optional, Sequence
44
+
45
+ # --- canon (ARCH-9C §2.1 CUA-loss term refs + §2.3 sub-attribution) ----------
46
+ # The browser-surface loss menu (the 9-tuple analogue of V1_IMAGE_LOSS_TERM_REFS,
47
+ # the 6-tuple). ``grounding_step_accuracy`` is admitted only under
48
+ # cua_surface=desktop — see V1_CUA_DESKTOP_ANCHOR_TERMS / unit 4.
49
+ V1_CUA_LOSS_TERM_REFS = (
50
+ "task_success",
51
+ "state_match",
52
+ "grounding_mutation_resilience",
53
+ "action_correctness",
54
+ "step_efficiency",
55
+ "safety_adherence",
56
+ "tool_evidence",
57
+ "trace_coverage",
58
+ "completion_judge",
59
+ )
60
+
61
+ # The MANDATORY deterministic post-state anchors (analogue of
62
+ # V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS). Both alias the SAME state_quality
63
+ # anchor (optimize.py:17565-17568 — url/final_state/order_id match); both admitted
64
+ # so authors can name the outcome either way. A browser-surface loss MUST carry
65
+ # >= 1 of these (unit 2). For desktop the admissible anchor narrows to
66
+ # V1_CUA_DESKTOP_ANCHOR_TERMS (unit 4).
67
+ V1_CUA_LOSS_DETERMINISTIC_ANCHOR_TERMS = ("task_success", "state_match")
68
+
69
+ # The narrower deterministic anchor admitted under cua_surface=desktop
70
+ # (ARCH §2.4 / 9C-A7b) — grounding/step accuracy, NOT full task success.
71
+ V1_CUA_DESKTOP_ANCHOR_TERMS = ("grounding_step_accuracy",)
72
+
73
+ # The bounded/guarded judge contributors (analogue of V1_IMAGE_LOSS_JUDGE_TERMS);
74
+ # keyed (unit 4). A judge-only loss (terms subset of this set) is structurally
75
+ # rejected (unit 2 / 9C-D3).
76
+ V1_CUA_LOSS_JUDGE_TERMS = ("completion_judge",)
77
+
78
+ # The dual-objective safety axis (ST-WebAgentBench CuP / HarmonyGuard); the
79
+ # unsafe-completion canary rides it (unit 2). CuP-proxy:
80
+ # prompt_injection_touched==0 AND blocked_action==0 (the kit's action_quality
81
+ # safety half, optimize.py:17552-17554).
82
+ V1_CUA_LOSS_MANDATORY_SAFETY_TERMS = ("safety_adherence",)
83
+
84
+ # The four-token CUA sub-attribution closed set (9C §2.3), stamped alongside the
85
+ # base FAILURE_LAYERS tag (analogue of V1_IMAGE_FAILURE_SUBLAYERS).
86
+ V1_CUA_FAILURE_SUBLAYERS = ("perception", "grounding", "action_policy", "reasoning_memory")
87
+
88
+ # The cua_surface switch on WorldSpec.spec (analogue of V1_IMAGE_TASK_MODES). ONE
89
+ # world loop, two surface profiles (9C-D2): browser -> world.kind=browser (full
90
+ # post-state); desktop -> world.kind=computer_use (grounding/step, unit 4).
91
+ V1_CUA_SURFACES = ("browser", "desktop")
92
+
93
+ # A MARKER field on artifact metadata — NOT a new evidence class (R5/A18; the
94
+ # frozen EVIDENCE_CLASSES 4-tuple live/_contract.py:18 is unchanged). The analogue
95
+ # of V1_IMAGE_FIDELITY_TIERS. (ARCH §2.6)
96
+ V1_CUA_FIDELITY_TIERS = ("deterministic_fixture", "keyed_live_model")
97
+
98
+ # The typed ``kind`` discriminators a completion guard row may carry, beyond the
99
+ # base sentinel/canary rows the loss guard already allows (ARCH §2.2; the analogue
100
+ # of V1_IMAGE_PERCEPTION_GUARD_KINDS).
101
+ V1_CUA_COMPLETION_GUARD_KINDS = ("fake_completion", "unsafe_completion")
102
+
103
+ # NAMING MIRROR ONLY (9C-A1c / 9C-D4). References the kit's EXISTING mutation-pack
104
+ # operators (normalize_browser_mutation_pack, environment.py:5146;
105
+ # _browser_mutation_perturbations, :28727; the prompt-injection surfaces,
106
+ # :29350/:2903). There is NO cua_perturb.py and NO apply_cua_perturbations
107
+ # function — the sharpest contrast with 9B's image_perturb.py.
108
+ V1_CUA_PERTURBATION_OPERATORS = ("selector_drift", "layout_shift", "stale_screenshot", "injected_dom")
109
+
110
+ # The registered world-kind tokens + the namespaced extension names (R4 hook). The
111
+ # CUA loop registers EXECUTABLE-LOOP status for two already-frozen kinds.
112
+ CUA_BROWSER_WORLD_KIND = "browser"
113
+ CUA_DESKTOP_WORLD_KIND = "computer_use"
114
+ CUA_BROWSER_EXTENSION_NAME = "agentlearning.browser_cua"
115
+ CUA_DESKTOP_EXTENSION_NAME = "agentlearning.computer_use_cua"
116
+
117
+ # The R4 rung -> evidence-class ladder (ARCH §2.6). The deterministic core is
118
+ # local_gate/captured_fixture; live_lane is added ONLY on the keyed lane record
119
+ # (unit 7), never the day-one deterministic record.
120
+ _CUA_RUNG_LADDER = {
121
+ "rung1": ["local_gate"],
122
+ "perturbed": ["live_stressed", "captured_fixture"],
123
+ "keyed_browser_vm": ["live_lane"],
124
+ }
125
+
126
+
127
+ class CuaLossCompositionError(ValueError):
128
+ """Raised when a CUA objective violates the 9C-A2 composition rule (the
129
+ ``cua_loss_guard_missing`` finding — a CUA specialization of
130
+ ``objective_guards_missing``). A ``ValueError`` subclass so callers can
131
+ ``except ValueError`` exactly as for ``ImageLossCompositionError`` /
132
+ ``VoiceLossCompositionError``."""
133
+
134
+
135
+ def _term_refs(objective: Mapping[str, Any]) -> list[str]:
136
+ """The objective's eval refs (read from ``evals`` — the loss.py schema; also
137
+ tolerant of a ``terms`` alias)."""
138
+ rows = objective.get("evals") or objective.get("terms") or []
139
+ return [
140
+ str(term.get("eval"))
141
+ for term in rows
142
+ if isinstance(term, Mapping) and term.get("eval")
143
+ ]
144
+
145
+
146
+ def _admissible_anchor_terms(cua_surface: str) -> tuple[str, ...]:
147
+ """The surface-admissible deterministic anchor set (the analogue of the image
148
+ ``_admissible_anchor_terms(task_mode)``): browser -> the full post-state
149
+ anchors; desktop -> the narrower grounding/step anchor (unit 4)."""
150
+ if cua_surface == "desktop":
151
+ return V1_CUA_DESKTOP_ANCHOR_TERMS
152
+ return V1_CUA_LOSS_DETERMINISTIC_ANCHOR_TERMS
153
+
154
+
155
+ def _admissible_term_refs(cua_surface: str) -> tuple[str, ...]:
156
+ """The surface-admissible loss-term menu: browser -> V1_CUA_LOSS_TERM_REFS;
157
+ desktop -> the narrower grounding/step anchor + the deterministic-composition
158
+ + safety + judge terms (unit 4). Desktop drops the browser-only post-state
159
+ anchors (task_success / state_match) since the credential-free desktop rung is
160
+ grounding/step ONLY, not full task success."""
161
+ if cua_surface == "desktop":
162
+ return (
163
+ V1_CUA_DESKTOP_ANCHOR_TERMS
164
+ + (
165
+ "grounding_mutation_resilience",
166
+ "action_correctness",
167
+ "step_efficiency",
168
+ "safety_adherence",
169
+ "tool_evidence",
170
+ "trace_coverage",
171
+ )
172
+ + V1_CUA_LOSS_JUDGE_TERMS
173
+ )
174
+ return V1_CUA_LOSS_TERM_REFS
175
+
176
+
177
+ def attribute_cua_sublayer(
178
+ *,
179
+ failure_layer: str,
180
+ deficit: Mapping[str, Any] | None = None,
181
+ signal: str | None = None,
182
+ ) -> str:
183
+ """Map a weak CUA cell to a ``V1_CUA_FAILURE_SUBLAYERS`` token, stamped
184
+ ALONGSIDE the base ``FAILURE_LAYERS`` tag (a weak cell carries both, e.g.
185
+ ``{failure_layer:"agent_behavior", cua_sublayer:"grounding"}``). The base
186
+ attribution rides the existing ``practice/_diagnose.py`` machinery; this is the
187
+ thin sublayer helper (the CUA analogue of ``attribute_image_sublayer``,
188
+ image_loop.py:208).
189
+
190
+ Routing (ARCH-9C §2.3 table, grounded in the observe->ground->act decomposition
191
+ + the kit's layers ["browser","cua","security","evaluator"] optimize.py:17267 +
192
+ the step-level stuck/milestone split 2604.27151):
193
+ * stale screenshot, didn't refresh; missed an observed change -> ``perception``
194
+ (observation-channel failure);
195
+ * selector drifted, mis-clicked; coordinate off -> ``grounding`` (the
196
+ observe->ground seam, the dominant mutation-resilience failure);
197
+ * looped on the same step / 2.5-2.8x too many steps; touched injected banner
198
+ -> ``action_policy`` (action/escalation policy + safety);
199
+ * right perception, wrong plan; bad memory of prior steps ->
200
+ ``reasoning_memory`` (plan/memory failure — ACuRL / Reflexion)."""
201
+
202
+ sig = str(signal or (deficit or {}).get("signal") or "").lower()
203
+ # Precedence-ordered (the ARCH §2.3 routing table). reasoning_memory is
204
+ # checked FIRST among the higher-cognition cues so a "right perception, wrong
205
+ # plan; bad memory of prior steps" cell routes to reasoning_memory even though
206
+ # it mentions perception (the word is a red herring; the ARCH perception row is
207
+ # "stale screenshot / missed an observed change", which carries none of these
208
+ # cues).
209
+ if any(
210
+ k in sig
211
+ for k in ("wrong plan", "wrong_plan", "bad memory", "memory", "reflect",
212
+ "reasoning", "prior step", "prior_step")
213
+ ):
214
+ return "reasoning_memory"
215
+ if any(
216
+ k in sig
217
+ for k in ("stale screenshot", "stale_screenshot", "didn't refresh",
218
+ "did not refresh", "missed an observed", "missed_change",
219
+ "observed change", "screenshot")
220
+ ):
221
+ return "perception"
222
+ if any(
223
+ k in sig
224
+ for k in ("selector drift", "selector_drift", "drifted selector",
225
+ "selector drifted", "mis-click", "misclick", "mis click",
226
+ "coordinate off", "coordinate_off", "ground")
227
+ ):
228
+ return "grounding"
229
+ if any(
230
+ k in sig
231
+ for k in ("loop", "too many steps", "step_efficiency", "redundant",
232
+ "injected banner", "injected_banner", "injection", "blocked",
233
+ "escalation", "action_policy", "action policy", "unsafe")
234
+ ):
235
+ return "action_policy"
236
+ # default: infra-implicated cells land on perception (the cheapest observation
237
+ # fix before blaming the policy); otherwise the reasoning/memory layer.
238
+ if failure_layer in ("lane_infra", "framework_runtime", "provider"):
239
+ return "perception"
240
+ return "reasoning_memory"
241
+
242
+
243
+ def compile_cua_objective(
244
+ payload: Mapping[str, Any], *, cua_surface: str = "browser"
245
+ ) -> dict:
246
+ """Compile a multi-objective CUA loss with a fake/unsafe-completion Goodhart
247
+ guard (ARCH-9C §2.2 / 9C-A2 / 9C-D3). The CUA analogue of
248
+ ``compile_image_objective`` (image_loop.py:132). Enforces, ON TOP of the
249
+ verbatim ``loss.compile_objective`` Goodhart guard:
250
+
251
+ rule 1: closed-set ``cua_surface`` (browser | desktop);
252
+ rule 2: >= 2 terms (a single-term CUA objective is reward-hackable);
253
+ rule 3: >= 1 surface-admissible deterministic post-state anchor — a
254
+ judge-only loss is INVALID (9C-D3). ``cua_surface`` selects the
255
+ admissible anchor set: browser -> V1_CUA_LOSS_DETERMINISTIC_ANCHOR_TERMS;
256
+ desktop -> V1_CUA_DESKTOP_ANCHOR_TERMS (the narrower grounding/step
257
+ anchor, unit 4);
258
+ rule 4: unknown-ref rejection (every term must be a member of the surface
259
+ menu);
260
+ rule 5: when sentinel/canary rows carry a completion ``kind`` discriminator
261
+ it must be in V1_CUA_COMPLETION_GUARD_KINDS (the closed set).
262
+
263
+ Then delegates to ``loss.compile_objective`` VERBATIM — which unconditionally
264
+ enforces the populated guard block (sentinel_rows / canary_evals,
265
+ min_guard_count >= 1, "There is no override.")."""
266
+
267
+ from . import loss as _loss # downward facade import (legal; image_loop.py idiom)
268
+
269
+ # rule 1: closed-set cua_surface.
270
+ if cua_surface not in V1_CUA_SURFACES:
271
+ raise CuaLossCompositionError(
272
+ f"cua_loss_guard_missing: cua_surface {cua_surface!r} not in "
273
+ f"{V1_CUA_SURFACES}"
274
+ )
275
+
276
+ refs = _term_refs(payload)
277
+
278
+ # rule 2: >= 2 terms.
279
+ if len(refs) < 2:
280
+ raise CuaLossCompositionError(
281
+ "cua_loss_guard_missing: a CUA objective is reward-hackable as a "
282
+ "single term; it MUST be multi-objective (>= 2 terms). "
283
+ f"got {refs}"
284
+ )
285
+
286
+ # rule 3: >= 1 surface-admissible deterministic post-state anchor (judge-only
287
+ # REJECTED — 9C-D3).
288
+ anchors = _admissible_anchor_terms(cua_surface)
289
+ if not any(ref in anchors for ref in refs):
290
+ raise CuaLossCompositionError(
291
+ "cua_loss_guard_missing: a CUA loss MUST carry >= 1 deterministic "
292
+ f"post-state anchor {anchors}; a judge-only loss is INVALID by "
293
+ f"contract (9C-D3). got {refs} (cua_surface={cua_surface})"
294
+ )
295
+
296
+ # rule 4: unknown-ref rejection (surface-filtered).
297
+ allowed = _admissible_term_refs(cua_surface)
298
+ for ref in refs:
299
+ if ref not in allowed:
300
+ raise CuaLossCompositionError(
301
+ f"cua_loss_guard_missing: unknown CUA loss term {ref!r}; "
302
+ f"expected members of {allowed} (cua_surface={cua_surface})"
303
+ )
304
+
305
+ # rule 5: the fake/unsafe-completion guard rows ride the existing
306
+ # sentinel_rows/canary_evals with a typed ``kind`` discriminator (no new
307
+ # ObjectiveSpec field, ARCH-9C §2.2). When present it must be in the closed set
308
+ # (plus any untyped/base rows the loss guard already allows).
309
+ guards = payload.get("guards") or {}
310
+ for bucket in ("sentinel_rows", "canary_evals"):
311
+ for row in guards.get(bucket) or []:
312
+ if isinstance(row, Mapping):
313
+ kind = row.get("kind")
314
+ if kind is not None and kind not in V1_CUA_COMPLETION_GUARD_KINDS:
315
+ raise CuaLossCompositionError(
316
+ f"cua_loss_guard_missing: guard row kind {kind!r} not in "
317
+ f"{V1_CUA_COMPLETION_GUARD_KINDS}"
318
+ )
319
+
320
+ # the verbatim Goodhart guard (loss.py:106-116) — "There is no override."
321
+ return _loss.compile_objective(payload)
322
+
323
+
324
+ def _validate_cua_world_spec(spec: Mapping[str, Any]) -> None:
325
+ """The R4 ``spec_validator`` for the CUA world: validate the ``cua_surface``
326
+ switch (browser | desktop) on ``WorldSpec.spec`` (the analogue of
327
+ ``_validate_image_world_spec``, image_loop.py:293). Raises ValueError on an
328
+ unknown surface (the closed-set guard)."""
329
+
330
+ cua_surface = str((spec or {}).get("cua_surface", "browser"))
331
+ if cua_surface not in V1_CUA_SURFACES:
332
+ raise ValueError(
333
+ f"cua world.spec.cua_surface {cua_surface!r} not in {V1_CUA_SURFACES}"
334
+ )
335
+
336
+
337
+ def _ensure_cua_world_registered(cua_surface: str = "browser") -> None:
338
+ """Flip ``browser`` / ``computer_use`` from typed-only to EXECUTABLE-LOOP
339
+ status via the R4 hook (ARCH-9C §2.1 / §2.3 / 9C-D2 / 9C-A1b). Idempotent BY
340
+ VENDOR.NAME.
341
+
342
+ THE 9C-A1b RULE (binding): this does NOT use the verbatim image idempotence
343
+ guard (``if kind in resolved_world_kinds(): return``, image_loop.py:272),
344
+ because ``browser`` / ``computer_use`` are ALREADY in ``resolved_world_kinds()``
345
+ as frozen built-ins (contract.py:55) — that guard would short-circuit
346
+ IMMEDIATELY and never record the R4 executable-loop evidence (spec_validator +
347
+ rung_ladder + rung1_fixture_green). Instead it gates on the EXECUTABLE-LOOP
348
+ RECORD presence in ``_EXTRA_WORLD_KINDS`` (the additive R4 record keyed by
349
+ vendor.name). ``register_world_kind`` pushes the CUA record into
350
+ ``_EXTRA_WORLD_KINDS``; built-ins shadow extensions at resolution
351
+ (contract.py:88-91), so ``WorldSpec(kind="browser")`` keeps validating against
352
+ the built-in entry and the frozen ``SIMULATION_WORLD_KINDS`` tuple stays
353
+ byte-stable. ``browser`` / ``computer_use`` stay in
354
+ ``TYPED_ONLY_WORLD_KINDS_V1`` — executable-loop status is carried by the
355
+ registry record + the ``cua_loop_readiness`` gate, NOT by the frozen
356
+ executable tuple."""
357
+
358
+ from fi.simulate.simulation import contract as _contract
359
+
360
+ if cua_surface not in V1_CUA_SURFACES:
361
+ raise CuaLossCompositionError(
362
+ f"cua_loss_guard_missing: cua_surface {cua_surface!r} not in "
363
+ f"{V1_CUA_SURFACES}"
364
+ )
365
+
366
+ if cua_surface == "browser":
367
+ kind_token = CUA_BROWSER_WORLD_KIND
368
+ vendor_name = CUA_BROWSER_EXTENSION_NAME
369
+ else:
370
+ kind_token = CUA_DESKTOP_WORLD_KIND
371
+ vendor_name = CUA_DESKTOP_EXTENSION_NAME
372
+
373
+ from . import extensions as _ext
374
+
375
+ # 9C-A1b: gate on the EXECUTABLE-LOOP RECORD, not on bare admissibility. The
376
+ # kind is already admissible (built-in); the executable-loop marker is the
377
+ # _EXTRA_WORLD_KINDS record carrying the CUA kind_token + the vendor.name.
378
+ # register_extension keys _EXTRA_WORLD_KINDS by the kind_token (it calls
379
+ # contract.register_world_kind(token, stored)), so the contract record is read
380
+ # by kind_token; the vendor.name lives inside the record's ``name`` field
381
+ # (the idempotence key per 9C-A1b — one executable-loop record per vendor).
382
+ existing = _contract._EXTRA_WORLD_KINDS.get(kind_token) # additive record, never the built-in
383
+ if existing and existing.get("kind_token") == kind_token and existing.get("name") == vendor_name:
384
+ return # executable-loop record already present (idempotent by vendor.name)
385
+
386
+ # The persistent extension registry (extensions._REGISTRY) outlives the
387
+ # contract's _EXTRA_WORLD_KINDS within a process. register_extension RAISES on
388
+ # a name collision, so if the extension is ALREADY in the registry (e.g. the
389
+ # contract record was cleared but the registry was not), re-push the existing
390
+ # stored record into the contract directly rather than re-registering. This
391
+ # keeps the gate idempotent by vendor.name AND restores the executable-loop
392
+ # evidence in _EXTRA_WORLD_KINDS.
393
+ stored = _ext.resolve("environment", vendor_name)
394
+ if stored is not None and stored.get("kind_token") == kind_token:
395
+ _contract.register_world_kind(kind_token, stored)
396
+ return
397
+
398
+ _ext.register_extension(
399
+ "environment",
400
+ {
401
+ "name": vendor_name, # vendor.name shape (_validate_record)
402
+ "kind_token": kind_token, # "browser" / "computer_use" token
403
+ "spec_validator": _validate_cua_world_spec, # R4 mandate
404
+ "rung_ladder": _CUA_RUNG_LADDER, # R4 mandate
405
+ # the deterministic core; live_lane is added ONLY on the keyed lane
406
+ # record (unit 7), never here. gated_contexts_runnable stays False
407
+ # until rung1_fixture_green (extensions.py admission); 9C never
408
+ # silently claims executable.
409
+ "evidence_class_capability": ["local_gate", "captured_fixture"],
410
+ },
411
+ )
412
+
413
+
414
+ def build_cua_practice_loop_manifest(
415
+ *,
416
+ name: str,
417
+ base_agent: Mapping[str, Any],
418
+ search_space: Mapping[str, Sequence[Any]],
419
+ objective: Mapping[str, Any],
420
+ eval_budget: int,
421
+ seed: int,
422
+ cua_surface: str = "browser",
423
+ scenario_inline: Optional[Mapping[str, Any]] = None,
424
+ max_rounds: int = 8,
425
+ ) -> dict[str, Any]:
426
+ """Assemble the CUA improvement-loop manifest: the 13D Practice Loop on
427
+ ``world.kind=browser`` / ``computer_use`` + ``cua_surface`` with the
428
+ multi-objective guarded CUA loss + the whole CUA-agent search space (9C-D5).
429
+ Delegates to ``optimize.build_practice_loop_manifest`` so its validators hold
430
+ VERBATIM (9C-A3). The objective is compiled by ``compile_cua_objective`` (the
431
+ 9C-A2 rule) before it rides the simulation.
432
+
433
+ Byte-parallel to ``build_image_practice_loop_manifest`` except: (a) the
434
+ ``_ensure_cua_world_registered(cua_surface)`` call uses the 9C-A1b
435
+ executable-loop-record gate (NOT the verbatim image idempotence guard); (b)
436
+ ``world["kind"]`` is ``"browser"`` / ``"computer_use"`` driven by
437
+ ``cua_surface`` (image's is always ``"image"``); (c) the ``spec["cua_surface"]``
438
+ write (instead of ``spec["task_mode"]``)."""
439
+
440
+ from . import optimize as _optimize # downward facade import (legal)
441
+
442
+ if cua_surface not in V1_CUA_SURFACES:
443
+ raise CuaLossCompositionError(
444
+ f"cua_loss_guard_missing: cua_surface {cua_surface!r} not in "
445
+ f"{V1_CUA_SURFACES}"
446
+ )
447
+
448
+ _ensure_cua_world_registered(cua_surface) # step 1 (§3.1, 9C-A1b)
449
+ compiled = compile_cua_objective(objective, cua_surface=cua_surface) # step 2 (unit 2)
450
+ inline = dict(scenario_inline or {})
451
+ inline.setdefault("version", "agent-learning.simulation.v1")
452
+ inline["objective"] = compiled
453
+ world = dict(inline.get("world") or {})
454
+ world["kind"] = (
455
+ CUA_BROWSER_WORLD_KIND if cua_surface == "browser" else CUA_DESKTOP_WORLD_KIND
456
+ ) # step 3 — the registered kind
457
+ spec = dict(world.get("spec") or {})
458
+ spec["cua_surface"] = cua_surface # the cua_surface switch on WorldSpec.spec
459
+ world["spec"] = spec
460
+ inline["world"] = world
461
+
462
+ return _optimize.build_practice_loop_manifest( # step 4 — VERBATIM delegate
463
+ name=name,
464
+ simulation={"version": inline["version"], "inline": inline},
465
+ base_agent=base_agent,
466
+ search_space=search_space,
467
+ eval_budget=eval_budget,
468
+ seed=seed,
469
+ max_rounds=max_rounds,
470
+ )
471
+
472
+
473
+ # === Unit 7 — the keyed real-browser/VM lane (opt-in, NEVER a gate prerequisite) ===
474
+ # ARCH-9C §2.4 / §2.6 / 9C-D1/D6/A8. The keyed completion_judge term, the desktop
475
+ # full-post-state rungs, and the one real-browser/CUA-agent live-proof are
476
+ # owner-keyed/infra-provisioned, opt-in, never a release gate. The deterministic
477
+ # core stays local_gate/captured_fixture; the keyed lane is the ONLY honest place
478
+ # for live_lane (a real keyed browser/VM/judge ran).
479
+
480
+ KEYED_CUA_EXTENSION_NAME = "agentlearning.cua.keyed"
481
+
482
+ # the env keys that gate the keyed completion_judge lane (checked, never required
483
+ # by any gate). The GRADE-shaped judge term (9C-A8) calls a judge model.
484
+ CUA_JUDGE_KEY_ENVS = ("AGENT_LEARNING_CUA_JUDGE_KEY", "OPENAI_API_KEY")
485
+
486
+
487
+ class CuaKeyedLaneUnavailable(RuntimeError):
488
+ """Raised by the keyed lane when no judge key / VM infra is present — the loud
489
+ refusal (the ``cua_judge_key_unavailable`` / ``cua_desktop_infra_unavailable``
490
+ finding). The deterministic core NEVER raises this; only the opt-in keyed path
491
+ does."""
492
+
493
+
494
+ def cua_judge_key_present() -> bool:
495
+ """True iff a judge key is configured for the keyed completion_judge lane."""
496
+ import os
497
+
498
+ return any(os.environ.get(env) for env in CUA_JUDGE_KEY_ENVS)
499
+
500
+
501
+ def register_keyed_cua_lane() -> None:
502
+ """Register the SEPARATE keyed-lane extension record that adds ``live_lane`` to
503
+ ``evidence_class_capability`` (ARCH-9C §2.6, unit 7). Idempotent. This is the
504
+ ONLY record that may carry ``live_lane`` — the deterministic-core records
505
+ (``_ensure_cua_world_registered``) stay ``("local_gate","captured_fixture")``.
506
+ NEVER called by the gate; opt-in only."""
507
+ from . import extensions as _ext
508
+
509
+ if _ext.resolve("environment", KEYED_CUA_EXTENSION_NAME) is not None:
510
+ return
511
+ _ext.register_extension(
512
+ "environment",
513
+ {
514
+ "name": KEYED_CUA_EXTENSION_NAME,
515
+ # the keyed lane declares the keyed capability without a second
516
+ # kind_token (the world kinds are already registered).
517
+ "evidence_class_capability": ["local_gate", "captured_fixture", "live_lane"],
518
+ },
519
+ )
520
+
521
+
522
+ def run_keyed_cua_live_proof(
523
+ *,
524
+ base_agent: Mapping[str, Any],
525
+ search_space: Mapping[str, Sequence[Any]],
526
+ objective: Mapping[str, Any],
527
+ eval_budget: int,
528
+ seed: int,
529
+ cua_surface: str = "browser",
530
+ name: str = "cua-keyed-live-proof",
531
+ ) -> dict[str, Any]:
532
+ """The one owner-keyed live-proof entry (WORKFLOW Step 5 real-keys ground
533
+ rule). Refuses LOUDLY without a key (``CuaKeyedLaneUnavailable`` ->
534
+ ``cua_judge_key_unavailable``) — never a fake number, never a release
535
+ prerequisite. With a key, it builds the manifest and marks the run
536
+ ``live_lane`` / ``fidelity_tier=keyed_live_model``.
537
+
538
+ The keyed run itself (calling the GRADE judge / a real browser / a VM) is left
539
+ to the caller's runtime; this returns the keyed manifest + the honest
540
+ evidence-class stamp so an owner can execute it once with real keys/infra."""
541
+ if not cua_judge_key_present():
542
+ raise CuaKeyedLaneUnavailable(
543
+ "cua_judge_key_unavailable: the keyed real-browser/VM lane requires a "
544
+ f"judge key (one of {CUA_JUDGE_KEY_ENVS}); withheld -- never a fake "
545
+ "number, never a release prerequisite"
546
+ )
547
+ register_keyed_cua_lane()
548
+ manifest = build_cua_practice_loop_manifest(
549
+ name=name,
550
+ base_agent=base_agent,
551
+ search_space=search_space,
552
+ objective=objective,
553
+ eval_budget=eval_budget,
554
+ seed=seed,
555
+ cua_surface=cua_surface,
556
+ )
557
+ return {
558
+ "manifest": manifest,
559
+ "evidence_class": "live_lane", # the ONLY honest live_lane
560
+ "fidelity_tier": "keyed_live_model",
561
+ "cua_surface": cua_surface,
562
+ }