agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,241 @@
1
+ """Phase 9B unit 1b — pure-numpy seeded image perturbation operators (the image
2
+ analogue of 9A's ``live/_perturb.py`` acoustic operators).
3
+
4
+ ARCH-9B §2.1 / decision 9B-A1b (companion module home — substrate not loop),
5
+ 9B-A6 (PURE-NUMPY v1, ZERO new dep — settles Open Q4).
6
+
7
+ MANDATORY (9B-A6): imports are **numpy + stdlib ONLY**. There is NO Pillow, no
8
+ scipy, no cv2, no imageio, no scikit-image — verified ``pyproject.toml`` carries
9
+ only ``numpy>=1.26.4``. Adding Pillow for the perturbation set would be a NEW
10
+ dependency + a license-audit obligation on the public repo's Apache-2.0 posture.
11
+ The kit's live substrate already imports numpy directly. A true-libjpeg or
12
+ PNG-render path is a NAMED post-v1 Pillow extra, auto-skip when absent, never a
13
+ v1 gate dependency.
14
+
15
+ Operators are deterministic under a recorded seed so stressed runs replay
16
+ byte-identically (the determinism the gate re-asserts). Each operates on a numpy
17
+ ``uint8`` raster (H x W x C) and is computed as a paired clean-vs-stressed delta
18
+ (the ``_perturb.apply_text_perturbations`` discipline).
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import Any, Mapping, Sequence
24
+
25
+ import numpy as np
26
+
27
+ # closed set; analogue of _perturb.py PERTURBATION_OPERATORS. All pure-numpy, all
28
+ # deterministic-under-seed. apply_image_perturbations RAISES for any operator not
29
+ # in this set (the _perturb.py raise-wall pattern generalized).
30
+ V1_IMAGE_PERTURBATION_OPERATORS = ("blur", "jpeg_compress", "resolution_drop", "occlusion")
31
+
32
+
33
+ class ImagePerturbationError(ValueError):
34
+ """Raised for an unknown operator or a mis-shaped raster (a contract error —
35
+ the _perturb.py raise-wall analogue). A ``ValueError`` subclass."""
36
+
37
+
38
+ def _require_raster(raster: Any, *, where: str) -> np.ndarray:
39
+ """Type-guard the input as a numpy ``uint8`` H x W x C raster. A non-uint8 /
40
+ non-3D input raises ``ImagePerturbationError`` — we never silently
41
+ mis-shape."""
42
+
43
+ if not isinstance(raster, np.ndarray):
44
+ raise ImagePerturbationError(
45
+ f"{where} needs a numpy uint8 H x W x C raster; got {type(raster).__name__}"
46
+ )
47
+ if raster.ndim != 3:
48
+ raise ImagePerturbationError(
49
+ f"{where} needs a 3-D (H x W x C) raster; got ndim={raster.ndim}"
50
+ )
51
+ if raster.dtype != np.uint8:
52
+ raise ImagePerturbationError(
53
+ f"{where} needs a uint8 raster; got dtype={raster.dtype}"
54
+ )
55
+ return raster
56
+
57
+
58
+ def blur(raster: np.ndarray, *, kernel_radius: int = 1, seed: int = 0) -> np.ndarray:
59
+ """Separable box-kernel blur as a numpy stride convolution (no scipy). A box
60
+ average over a ``(2*kernel_radius+1)`` window, applied separably across rows
61
+ then columns with edge replication. Deterministic (no rng draw); the ``seed``
62
+ is accepted for a uniform operator signature."""
63
+
64
+ arr = _require_raster(raster, where="blur").astype(np.float64)
65
+ radius = max(int(kernel_radius), 0)
66
+ if radius == 0:
67
+ return arr.astype(np.uint8)
68
+ width = 2 * radius + 1
69
+
70
+ def _box_axis(data: np.ndarray, axis: int) -> np.ndarray:
71
+ padded = np.pad(
72
+ data,
73
+ [(radius, radius) if a == axis else (0, 0) for a in range(data.ndim)],
74
+ mode="edge",
75
+ )
76
+ acc = np.zeros_like(data)
77
+ for offset in range(width):
78
+ sl = [slice(None)] * data.ndim
79
+ sl[axis] = slice(offset, offset + data.shape[axis])
80
+ acc = acc + padded[tuple(sl)]
81
+ return acc / float(width)
82
+
83
+ out = _box_axis(_box_axis(arr, 0), 1)
84
+ return np.clip(np.rint(out), 0, 255).astype(np.uint8)
85
+
86
+
87
+ def jpeg_compress(raster: np.ndarray, *, quality: int = 50, seed: int = 0) -> np.ndarray:
88
+ """Block-DCT quantization approximation in pure numpy (8x8 DCT-II matrices +
89
+ a quality-keyed quant table). A true libjpeg path is the post-v1 Pillow extra
90
+ (auto-skip). Deterministic (no rng draw); ``seed`` accepted for a uniform
91
+ signature."""
92
+
93
+ arr = _require_raster(raster, where="jpeg_compress").astype(np.float64)
94
+ q = int(np.clip(quality, 1, 100))
95
+ # the standard JPEG quality -> scale heuristic.
96
+ if q < 50:
97
+ scale = 5000.0 / q
98
+ else:
99
+ scale = 200.0 - 2.0 * q
100
+ quant = max(1.0, scale / 16.0) # a single flat quantization step (luma-ish)
101
+
102
+ n = 8
103
+ k = np.arange(n)
104
+ # DCT-II orthonormal basis (8x8), built deterministically.
105
+ basis = np.cos(np.pi * (2 * k[:, None] + 1) * k[None, :] / (2 * n))
106
+ basis *= np.sqrt(2.0 / n)
107
+ basis[0, :] = np.sqrt(1.0 / n)
108
+ # basis[i, x] applies the i-th cosine over sample x; forward = basis @ block.
109
+
110
+ h, w, c = arr.shape
111
+ pad_h = (-h) % n
112
+ pad_w = (-w) % n
113
+ padded = np.pad(arr, ((0, pad_h), (0, pad_w), (0, 0)), mode="edge")
114
+
115
+ out = np.empty_like(padded)
116
+ for ch in range(c):
117
+ plane = padded[:, :, ch] - 128.0
118
+ for r0 in range(0, padded.shape[0], n):
119
+ for c0 in range(0, padded.shape[1], n):
120
+ block = plane[r0:r0 + n, c0:c0 + n]
121
+ coeffs = basis @ block @ basis.T
122
+ quantized = np.round(coeffs / quant) * quant
123
+ restored = basis.T @ quantized @ basis
124
+ out[r0:r0 + n, c0:c0 + n, ch] = restored + 128.0
125
+
126
+ out = out[:h, :w, :]
127
+ return np.clip(np.rint(out), 0, 255).astype(np.uint8)
128
+
129
+
130
+ def resolution_drop(raster: np.ndarray, *, scale: float = 0.5, seed: int = 0) -> np.ndarray:
131
+ """numpy decimate -> upsample (nearest), the band-limit analogue (the
132
+ ``resample_8k`` analogue from voice). Downscale by ``scale`` then nearest-
133
+ neighbour back to the original shape, destroying high-frequency detail.
134
+ Deterministic (no rng draw); ``seed`` accepted for a uniform signature."""
135
+
136
+ arr = _require_raster(raster, where="resolution_drop")
137
+ s = float(scale)
138
+ if not 0.0 < s < 1.0:
139
+ # scale outside (0,1) is a no-op (full resolution).
140
+ return arr.copy()
141
+ h, w, _ = arr.shape
142
+ small_h = max(1, int(round(h * s)))
143
+ small_w = max(1, int(round(w * s)))
144
+ # deterministic nearest-neighbour decimation.
145
+ row_idx = (np.arange(small_h) * (h / small_h)).astype(np.int64)
146
+ col_idx = (np.arange(small_w) * (w / small_w)).astype(np.int64)
147
+ small = arr[np.ix_(row_idx, col_idx, np.arange(arr.shape[2]))]
148
+ # nearest-neighbour upsample back to (h, w).
149
+ up_rows = (np.arange(h) * (small_h / h)).astype(np.int64)
150
+ up_cols = (np.arange(w) * (small_w / w)).astype(np.int64)
151
+ up = small[np.ix_(up_rows, up_cols, np.arange(arr.shape[2]))]
152
+ return up.astype(np.uint8)
153
+
154
+
155
+ def occlusion(raster: np.ndarray, *, coverage: float = 0.2, seed: int = 0) -> np.ndarray:
156
+ """Seeded rectangular mask zeroing a region (``np.random.default_rng(seed)``).
157
+ The mask covers approximately ``coverage`` of the area; its position is keyed
158
+ on the seed so a re-run is byte-identical."""
159
+
160
+ arr = _require_raster(raster, where="occlusion").copy()
161
+ cov = float(np.clip(coverage, 0.0, 1.0))
162
+ if cov <= 0.0:
163
+ return arr
164
+ h, w, _ = arr.shape
165
+ rng = np.random.default_rng(seed)
166
+ side = float(np.sqrt(cov))
167
+ box_h = max(1, int(round(h * side)))
168
+ box_w = max(1, int(round(w * side)))
169
+ top = int(rng.integers(0, max(1, h - box_h + 1)))
170
+ left = int(rng.integers(0, max(1, w - box_w + 1)))
171
+ arr[top:top + box_h, left:left + box_w, :] = 0
172
+ return arr
173
+
174
+
175
+ _OPERATOR_FNS = {
176
+ "blur": blur,
177
+ "jpeg_compress": jpeg_compress,
178
+ "resolution_drop": resolution_drop,
179
+ "occlusion": occlusion,
180
+ }
181
+
182
+
183
+ def perturbations_stanza(
184
+ applied: Sequence[Mapping[str, Any]],
185
+ *,
186
+ seed: int,
187
+ paired_clean_run: str | None = None,
188
+ ) -> dict[str, Any]:
189
+ """The applied-operator stanza (the ``_perturb.perturbations_stanza``
190
+ analogue): operator list, recorded seed, and the clean-twin link (deltas
191
+ render upstream)."""
192
+
193
+ return {
194
+ "operators": [dict(record) for record in applied],
195
+ "seed": seed,
196
+ "paired_clean_run": paired_clean_run,
197
+ }
198
+
199
+
200
+ def apply_image_perturbations(
201
+ raster: np.ndarray,
202
+ *,
203
+ operators: Sequence[str],
204
+ seed: int = 0,
205
+ params: Mapping[str, Any] | None = None,
206
+ paired_clean_run: str | None = None,
207
+ ) -> dict[str, Any]:
208
+ """Walk the operator list applying each with ``seed + index`` (the
209
+ ``_perturb.apply_text_perturbations`` pattern). Returns
210
+ ``{"raster": np.ndarray, "stanza": {...}, "paired_clean_run": <ref>}``.
211
+
212
+ The stanza mirrors ``perturbations_stanza`` — the applied-operator list + the
213
+ ``paired_clean_run`` link. The ``WorldSpec.perturbation_profile`` field
214
+ (contract.py:214) carries the profile LABEL on the stressed run.
215
+
216
+ RAISES ``ImagePerturbationError`` for any operator not in
217
+ ``V1_IMAGE_PERTURBATION_OPERATORS`` (a contract error — the raise-wall).
218
+
219
+ DETERMINISM (the gate asserts this, unit 5): same raster + same operators +
220
+ same seed => byte-identical output raster. No wall-clock, no randomness
221
+ outside the keyed rng."""
222
+
223
+ out = _require_raster(raster, where="apply_image_perturbations").copy()
224
+ params = dict(params or {})
225
+ applied: list[dict[str, Any]] = []
226
+ for index, operator in enumerate(operators):
227
+ if operator not in V1_IMAGE_PERTURBATION_OPERATORS:
228
+ raise ImagePerturbationError(
229
+ f"unknown perturbation operator {operator!r}; "
230
+ f"expected one of {V1_IMAGE_PERTURBATION_OPERATORS}"
231
+ )
232
+ op_seed = seed + index
233
+ op_params = dict(params.get(operator) or {})
234
+ out = _OPERATOR_FNS[operator](out, seed=op_seed, **op_params)
235
+ applied.append({"operator": operator, "seed": op_seed, **op_params})
236
+
237
+ return {
238
+ "raster": out,
239
+ "stanza": perturbations_stanza(applied, seed=seed, paired_clean_run=paired_clean_run),
240
+ "paired_clean_run": paired_clean_run,
241
+ }
fi/alk/improve.py ADDED
@@ -0,0 +1,274 @@
1
+ """Code-level RSI — fix a framework agent's actual CODE by run→trace→diagnose→
2
+ patch→re-run→keep-if-better-on-held-out.
3
+
4
+ The general self-improvement model: the `update` ACTION is a CODE EDIT (not a
5
+ config patch). The loop runs the framework agent's real source in sim, reads the
6
+ trace + the (discriminating, objective-anchored) eval, asks a proposer to PATCH
7
+ the source to fix the failure, applies the patch in an ISOLATED workdir, re-runs
8
+ on HELD-OUT tasks, and accepts ONLY if held-out improves AND a regression split
9
+ does not drop (no-forgetting). Reuses the proven run→eval foundation
10
+ (run_benchmark + objective_score); the genuinely new surface is exactly two
11
+ things — propose-a-code-edit and sandboxed re-run — both isolated here.
12
+
13
+ SAFETY (the code_exec verdict-#4 line holds): a code patch is LLM-AUTHORED code
14
+ run automatically in a loop. The patched source is written to an isolated temp
15
+ workdir and the agent is pointed at it by ABSOLUTE path — a bad mutation NEVER
16
+ touches the caller's real file unless the caller explicitly writes back an
17
+ accepted patch. Each candidate run is wall-clock bounded. This is for optimizing
18
+ a TRUSTED agent's own source; arbitrary untrusted code execution remains the
19
+ parked code_exec concern (needs a real sandbox).
20
+
21
+ OVERCLAIM GUARD: "improved" = held-out bug-class tasks pass on the DETERMINISTIC
22
+ anchor (e.g. tool_calls / completion_without_effort) the loop never optimized
23
+ directly, AND the regression split is not worse. A code-RSI loop is the most
24
+ gameable thing in the kit — never accept on the metric it edited toward.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import tempfile
30
+ from pathlib import Path
31
+ from typing import Any, Callable, Mapping, Sequence
32
+
33
+ from .tasks import run_benchmark
34
+
35
+ AGENT_LEARNING_CODE_RSI_REPORT_KIND = "agent-learning.code-rsi-report.v1"
36
+
37
+ # a proposer maps a diagnosis -> new full source text (or None to give up).
38
+ PatchProposer = Callable[[Mapping[str, Any]], "str | None"]
39
+
40
+
41
+ def _agent_for_source(source_path: Path, symbol: str) -> dict[str, Any]:
42
+ """A python-callable agent pointed at a source file by ABSOLUTE path (so the
43
+ patched copy loads, never the caller's original)."""
44
+ return {"type": "python", "callable": f"{source_path.resolve()}:{symbol}"}
45
+
46
+
47
+ def _score_split(
48
+ source_path: Path,
49
+ symbol: str,
50
+ dataset: Mapping[str, Any],
51
+ split: str | None,
52
+ *,
53
+ seed: int,
54
+ runner: Any = None,
55
+ ) -> dict[str, Any]:
56
+ agent = _agent_for_source(source_path, symbol)
57
+ # detector-aware: a candidate that GAMES the scorer (claims completion with no
58
+ # tool calls on a tool-anchored objective) is FAILED — the deterministic
59
+ # anti-gaming anchor the objective alone misses. This is what makes the
60
+ # no-tool bug detectable and prevents the loop accepting a reward-hacking patch.
61
+ # emit_telemetry=False: the RSI loop emits ONE run (in improve_agent_code),
62
+ # not one per split-score across rounds (Phase 14).
63
+ res = run_benchmark(dataset, agent, split=split, seed=seed,
64
+ evidence_class="captured_fixture", detect_reward_hacks=True,
65
+ runner=runner, emit_telemetry=False)
66
+ return res["aggregate"], res["per_task"]
67
+
68
+
69
+ def _failing(per_task: Sequence[Mapping[str, Any]]) -> list[dict]:
70
+ return [dict(r) for r in per_task if r.get("verdict") != "pass"]
71
+
72
+
73
+ def _write_source(workdir: Path, version: int, text: str) -> Path:
74
+ p = workdir / f"agent_v{version}.py"
75
+ p.write_text(text, encoding="utf-8")
76
+ return p
77
+
78
+
79
+ def improve_agent_code(
80
+ *,
81
+ source_text: str,
82
+ symbol: str,
83
+ dataset: Mapping[str, Any],
84
+ propose_patch: PatchProposer,
85
+ objective: Mapping[str, Any],
86
+ train_split: str = "train",
87
+ test_split: str = "test",
88
+ regression_split: str | None = "regression",
89
+ max_rounds: int = 3,
90
+ threshold: float = 0.5,
91
+ seed: int = 42,
92
+ runner: Any = None,
93
+ emit_telemetry: bool = True,
94
+ project_name: str | None = None,
95
+ ) -> dict[str, Any]:
96
+ """Run the code-level RSI loop on ``source_text`` (a module defining ``symbol``)
97
+ against ``dataset`` (needs ``train``/``test`` splits; ``regression`` optional).
98
+
99
+ Returns a report: baseline vs accepted held-out scores, the accepted patch (or
100
+ None), per-round attempts, and the no-forgetting (regression) result. The
101
+ caller decides whether to write an accepted patch back to the real file."""
102
+
103
+ splits = dataset.get("splits") or {}
104
+ if not splits.get(train_split) or not splits.get(test_split):
105
+ raise ValueError("dataset needs both train and test splits for code-RSI")
106
+ has_regression = bool(regression_split and splits.get(regression_split))
107
+
108
+ with tempfile.TemporaryDirectory(prefix="agent-code-rsi-") as tmp:
109
+ workdir = Path(tmp)
110
+ cur = _write_source(workdir, 0, source_text)
111
+
112
+ base_test, _ = _score_split(cur, symbol, dataset, test_split, seed=seed, runner=runner)
113
+ base_reg = None
114
+ if has_regression:
115
+ base_reg, _ = _score_split(cur, symbol, dataset, regression_split, seed=seed, runner=runner)
116
+
117
+ rounds: list[dict[str, Any]] = []
118
+ accepted_text: str | None = None
119
+ cur_text = source_text
120
+ prior_attempts: list[dict[str, Any]] = [] # fed back so the loop LEARNS
121
+
122
+ train_agg, train_per = _score_split(cur, symbol, dataset, train_split, seed=seed, runner=runner)
123
+ for rnd in range(max_rounds):
124
+ failing = _failing(train_per)
125
+ if not failing:
126
+ rounds.append({"round": rnd, "status": "no_bug_on_train", "train_pass_rate": train_agg["pass_rate"]})
127
+ break
128
+
129
+ diagnosis = {
130
+ "current_source": cur_text,
131
+ "symbol": symbol,
132
+ "objective": objective,
133
+ "failing_examples": [
134
+ {"task_id": f.get("task_id"), "score": f.get("score"),
135
+ "metric_averages": f.get("metric_averages"),
136
+ "tool_calls": len(f.get("tool_calls") or []),
137
+ "rewardhack": f.get("rewardhack"), "error": f.get("error")}
138
+ for f in failing
139
+ ],
140
+ # the RSI signal: prior rejected patches + WHY (execution errors /
141
+ # no-lift), so the next proposal does not repeat the mistake.
142
+ "prior_attempts": prior_attempts,
143
+ "signal": "failing tasks; check tool use / completion-without-effort",
144
+ }
145
+ new_text = propose_patch(diagnosis)
146
+ if not new_text or new_text == cur_text:
147
+ rounds.append({"round": rnd, "status": "no_patch_proposed"})
148
+ break
149
+
150
+ cand = _write_source(workdir, rnd + 1, new_text)
151
+ cand_train, cand_train_per = _score_split(cand, symbol, dataset, train_split, seed=seed, runner=runner)
152
+ cand_test, _ = _score_split(cand, symbol, dataset, test_split, seed=seed, runner=runner)
153
+ cand_reg = None
154
+ if has_regression:
155
+ cand_reg, _ = _score_split(cand, symbol, dataset, regression_split, seed=seed, runner=runner)
156
+
157
+ held_out_lift = round(cand_test["mean_score"] - base_test["mean_score"], 6)
158
+ regression_ok = (not has_regression) or (cand_reg["mean_score"] >= base_reg["mean_score"] - 1e-9)
159
+ accept = held_out_lift > 0 and regression_ok
160
+ cand_errors = [r.get("error") for r in cand_train_per if r.get("error")]
161
+
162
+ rounds.append({
163
+ "round": rnd, "status": "accepted" if accept else "rejected",
164
+ "train_lift": round(cand_train["mean_score"] - train_agg["mean_score"], 6),
165
+ "held_out_lift": held_out_lift,
166
+ "regression_ok": regression_ok,
167
+ "held_out_baseline": base_test["mean_score"],
168
+ "held_out_candidate": cand_test["mean_score"],
169
+ "candidate_errors": cand_errors[:2],
170
+ })
171
+ if accept:
172
+ accepted_text = new_text
173
+ cur, cur_text = cand, new_text
174
+ base_test = cand_test
175
+ if has_regression:
176
+ base_reg = cand_reg
177
+ break # one accepted fix per call (vertical); caller can re-invoke
178
+ # rejected: feed this attempt (source + why it failed) back to the proposer.
179
+ prior_attempts.append({
180
+ "patch_excerpt": new_text[:500],
181
+ "execution_errors": cand_errors[:2],
182
+ "held_out_lift": held_out_lift,
183
+ "reason": ("crashed: " + str(cand_errors[0])) if cand_errors else "no held-out improvement",
184
+ })
185
+ cur, cur_text, train_agg, train_per = cand, new_text, cand_train, cand_train_per
186
+
187
+ report = {
188
+ "kind": AGENT_LEARNING_CODE_RSI_REPORT_KIND,
189
+ "fixed": accepted_text is not None,
190
+ "accepted_source": accepted_text,
191
+ "held_out_baseline": base_test["mean_score"] if accepted_text is None else rounds[-1]["held_out_baseline"],
192
+ "held_out_final": base_test["mean_score"],
193
+ "regression_held": (not has_regression) or all(
194
+ r.get("regression_ok", True) for r in rounds if r["status"] in ("accepted", "rejected")
195
+ ),
196
+ "rounds": rounds,
197
+ }
198
+ if emit_telemetry:
199
+ # ONE dashboard run for the whole code-RSI loop: root + per-round spans
200
+ # (P14). Side-channel; never alters the report.
201
+ from .telemetry import emit_run
202
+
203
+ lift = round(report["held_out_final"] - report["held_out_baseline"], 6)
204
+ summary = emit_run(
205
+ kind="code-rsi",
206
+ name=symbol,
207
+ metrics={
208
+ "fixed": report["fixed"],
209
+ "held_out_baseline": report["held_out_baseline"],
210
+ "held_out_final": report["held_out_final"],
211
+ "held_out_lift": lift,
212
+ "regression_held": report["regression_held"],
213
+ },
214
+ verdict="pass" if report["fixed"] and lift > 0 else "fail",
215
+ children=[
216
+ (
217
+ f"round:{r['round']}",
218
+ {"status": r.get("status"),
219
+ "held_out_lift": r.get("held_out_lift"),
220
+ "regression_ok": r.get("regression_ok")},
221
+ )
222
+ for r in rounds
223
+ ],
224
+ project_name=project_name,
225
+ )
226
+ report["telemetry"] = summary.as_dict()
227
+ return report
228
+
229
+
230
+ def propose_patch_via_llm(model: str = "gpt-4o-mini") -> PatchProposer:
231
+ """Default proposer: an LLM rewrites the source to fix the diagnosed failure.
232
+ Conditioned on the current source + failing eval + objective; returns the new
233
+ full source (no co-authoring of the fix — the model derives it from the
234
+ trace/eval). Keyed (litellm); credential-free tests use a deterministic
235
+ proposer instead."""
236
+
237
+ def _propose(diagnosis: Mapping[str, Any]) -> str | None:
238
+ import re
239
+
240
+ import litellm
241
+
242
+ prior = diagnosis.get("prior_attempts") or []
243
+ prior_block = (
244
+ "\n\n=== YOUR PRIOR REJECTED ATTEMPTS (do NOT repeat these mistakes) ===\n"
245
+ + str(prior)[:800]
246
+ if prior else ""
247
+ )
248
+ prompt = (
249
+ "You are fixing a Python agent's source code. The agent runs in a "
250
+ "simulation. Each available tool is on `agent_input.tools` as a dict "
251
+ "shaped EITHER {\"name\": str, ...} OR {\"type\":\"function\",\"function\":{\"name\":str}} "
252
+ "— there is NO 'id' key on a tool spec, so read the name defensively "
253
+ "(`t.get('name') or (t.get('function') or {}).get('name')`) and generate "
254
+ "your own call id. The function MUST return a dict "
255
+ "{\"content\": str, \"tool_calls\": [{\"id\": str, \"name\": str, \"arguments\": dict}]}. "
256
+ "It is FAILING because it does not call the available tool (it fabricates "
257
+ "an answer). Rewrite the WHOLE source so it calls the first available "
258
+ "tool by its resolved name with empty arguments and returns that "
259
+ "tool_call. Keep the same function name. Code must run without KeyError. "
260
+ "Return ONLY the new Python source — no prose, no markdown fences."
261
+ + "\n\n=== CURRENT SOURCE ===\n"
262
+ + str(diagnosis.get("current_source", ""))
263
+ + "\n\n=== FAILING EVAL (sample) ===\n"
264
+ + str(diagnosis.get("failing_examples", []))[:600]
265
+ + prior_block
266
+ )
267
+ resp = litellm.completion(
268
+ model=model, messages=[{"role": "user", "content": prompt}], max_tokens=600,
269
+ )
270
+ text = resp.choices[0].message.content or ""
271
+ text = re.sub(r"^```(?:python)?\n|\n```$", "", text.strip()) # strip fences if any
272
+ return text or None
273
+
274
+ return _propose
@@ -0,0 +1,154 @@
1
+ """Opt-in live framework lanes (Phase 3) — facade only.
2
+
3
+ Imports NOTHING framework-side: the substrate (contract/runner/transcript/
4
+ stats/attribution/capture) is stdlib+numpy by construction, lane modules
5
+ import frameworks lazily inside function bodies, and ``_workers/`` entry
6
+ modules only ever run as scrubbed-env subprocesses (P3-D1). Lanes are extras
7
+ + env-gated markers, NEVER release prerequisites: every lane entry refuses
8
+ without its ``AGENT_LEARNING_LIVE_<LANE>=1`` flag.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import importlib
14
+ from typing import Any
15
+
16
+ _SUBMODULES = {
17
+ "_attribution",
18
+ "_capture",
19
+ "_codec",
20
+ "_contract",
21
+ "_loopback",
22
+ "_perturb",
23
+ "_runner",
24
+ "_stats",
25
+ "_transcript",
26
+ "a2a_lane",
27
+ "langgraph_lane",
28
+ "livekit_lane",
29
+ "mcp_lane",
30
+ "pipecat_lane",
31
+ "voice_redteam",
32
+ }
33
+
34
+ # lane name (LANE_ENV_FLAGS / LANE_EXTRAS key) → (module, entry point)
35
+ LANE_RUNNERS = {
36
+ "livekit": ("fi.alk.live.livekit_lane", "run_livekit_lane"),
37
+ "pipecat": ("fi.alk.live.pipecat_lane", "run_pipecat_lane"),
38
+ "langchain": ("fi.alk.live.langgraph_lane", "run_langgraph_lane"),
39
+ "mcp": ("fi.alk.live.mcp_lane", "run_mcp_lane"),
40
+ "a2a": ("fi.alk.live.a2a_lane", "run_a2a_lane"),
41
+ }
42
+
43
+ # public name → home module (resolved lazily so `import fi.alk.live`
44
+ # stays trivially cheap and provably framework-free)
45
+ _LAZY_EXPORTS = {
46
+ "AGENT_LEARNING_RUN_KIND": "_contract",
47
+ "EVIDENCE_CLASSES": "_contract",
48
+ "RELEASE_ADMISSIBLE_EVIDENCE_CLASSES": "_contract",
49
+ "FAILURE_LAYERS": "_contract",
50
+ "VERDICTS": "_contract",
51
+ "LANE_ENV_FLAGS": "_contract",
52
+ "LANE_EXTRAS": "_contract",
53
+ "LANE_BUDGET_S": "_contract",
54
+ "LANE_BUDGET_S_DEFAULT": "_contract",
55
+ "DEFAULT_REPEATS": "_contract",
56
+ "UNSTABLE_ICC_FLOOR": "_contract",
57
+ "LaneDisabledError": "_contract",
58
+ "LaneSpec": "_contract",
59
+ "LaneRun": "_contract",
60
+ "lane_budget_s": "_contract",
61
+ "require_lane_enabled": "_contract",
62
+ "LANE_SAFE_BASE_ENV": "_runner",
63
+ "LANE_BLOCKED_ENV": "_runner",
64
+ "LaneProcessResult": "_runner",
65
+ "scrubbed_lane_env": "_runner",
66
+ "spawn_lane_subprocess": "_runner",
67
+ "run_worker_once": "_runner",
68
+ "version_ok": "_runner",
69
+ "version_preflight": "_runner",
70
+ "TranscriptRecorder": "_transcript",
71
+ "read_transcript": "_transcript",
72
+ "redact_env_values": "_transcript",
73
+ "TRANSCRIPT_MAX_BYTES_ENV": "_transcript",
74
+ "LaneRunResult": "_stats",
75
+ "run_repeated": "_stats",
76
+ "lane_run_payload": "_stats",
77
+ "icc_and_within_variance": "_stats",
78
+ "divergence_step": "_stats",
79
+ "determinism_metrics": "_stats",
80
+ "derive_channel_evidence": "_stats",
81
+ "FailureAttribution": "_attribution",
82
+ "attribute_failure": "_attribution",
83
+ "CaptureRefusedError": "_capture",
84
+ "CAPTURE_PROVENANCE_FIELDS": "_capture",
85
+ "capture_to_fixture": "_capture",
86
+ "replay_fixture": "_capture",
87
+ "run_voice_escalation_campaign": "voice_redteam",
88
+ "compile_arc_turns": "voice_redteam",
89
+ "timing_fidelity": "voice_redteam",
90
+ "validate_authorization": "voice_redteam",
91
+ "VoiceAuthorizationError": "voice_redteam",
92
+ # Phase-12 12C rung-2: acoustic operators over the loopback PCM channel.
93
+ "apply_acoustic_perturbations": "_perturb",
94
+ "apply_reverb_blend": "_perturb",
95
+ "ACOUSTIC_RUNG_OPERATORS": "_perturb",
96
+ # Phase 9A: codec-survival facade (9A-A12, home _codec) + loopback runner
97
+ "score_codec_survival": "_codec",
98
+ "CodecUnsupportedError": "_codec",
99
+ "run_loopback_roundtrip": "_loopback",
100
+ "LoopbackFixtureMissing": "_loopback",
101
+ }
102
+
103
+
104
+ def run_lane(lane: str, *args: Any, **kwargs: Any) -> dict[str, Any]:
105
+ """Dispatch to a lane's entry point by name (the CLI front door's hook).
106
+
107
+ The lane module is imported lazily; its entry calls
108
+ ``require_lane_enabled`` first, so a missing env flag refuses before any
109
+ framework import is attempted.
110
+ """
111
+
112
+ try:
113
+ module_name, entry_name = LANE_RUNNERS[lane]
114
+ except KeyError:
115
+ known = ", ".join(sorted(LANE_RUNNERS))
116
+ raise ValueError(f"unknown live lane {lane!r}; expected one of: {known}")
117
+ module = importlib.import_module(module_name)
118
+ entry = getattr(module, entry_name)
119
+ return entry(*args, **kwargs)
120
+
121
+
122
+ def capture_fixture(*args: Any, **kwargs: Any) -> Any:
123
+ """Live→fixture demotion (see ``_capture.capture_to_fixture``)."""
124
+
125
+ from ._capture import capture_to_fixture as _capture_to_fixture
126
+
127
+ return _capture_to_fixture(*args, **kwargs)
128
+
129
+
130
+ def __getattr__(name: str) -> Any:
131
+ if name in _SUBMODULES:
132
+ module = importlib.import_module(f"{__name__}.{name}")
133
+ globals()[name] = module
134
+ return module
135
+ if name in _LAZY_EXPORTS:
136
+ module = importlib.import_module(
137
+ f"{__name__}.{_LAZY_EXPORTS[name]}"
138
+ )
139
+ value = getattr(module, name)
140
+ globals()[name] = value
141
+ return value
142
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
143
+
144
+
145
+ def __dir__() -> list[str]:
146
+ return sorted({*globals(), *_SUBMODULES, *_LAZY_EXPORTS})
147
+
148
+
149
+ __all__ = [
150
+ "LANE_RUNNERS",
151
+ "capture_fixture",
152
+ "run_lane",
153
+ *sorted(_LAZY_EXPORTS),
154
+ ]