agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,213 @@
1
+ """Code-tests verifier — run held-out tests against candidate code in isolation.
2
+
3
+ This is the coding-modality verifier for the bench harness. It implements the
4
+ trustworthiness rule every serious coding benchmark uses: **the oracle is held
5
+ out** — the held-out checks live in a separate file the candidate code never
6
+ imports or sees, and they are executed by a harness-written runner, not by the
7
+ candidate.
8
+
9
+ Sandboxes:
10
+ * ``subprocess`` (default) — a fresh interpreter in a throwaway tempdir, with a
11
+ scrubbed environment (no harness secrets) and a hard wall-clock timeout. This
12
+ is the sandbox used by the credential-free release gate, which only ever runs
13
+ trusted, shipped reference code. It is **not** a security boundary against
14
+ deliberately hostile code (no real filesystem/network isolation); for
15
+ untrusted agent output use the Docker lane (bench step 15E).
16
+ * ``docker`` — per-task container isolation with a no-network default
17
+ (bench step 15E).
18
+
19
+ The convention for a checks file: it defines one or more ``check_*`` callables
20
+ that import the candidate module (``import solution``) and ``assert`` the
21
+ expected behaviour. The harness discovers them, runs each, and reports per-check
22
+ pass/fail — so a candidate that no-ops, prints a fake "success" message, returns
23
+ wrong answers, or fails to define the entrypoint is failed deterministically.
24
+
25
+ THREAT-MODEL NOTE: the runner and the candidate share one process, so the
26
+ deterministic-failure guarantee covers *accidental* gaming, not an *adversarial*
27
+ candidate. A candidate that knows this runner's protocol could, during its import
28
+ body (which runs before ``check_*``), print a forged ``{"results": ...}`` line and
29
+ exit 0, or read the checks file to reflect expected values. Hardening that
30
+ (process/UID separation of the oracle + an authenticated out-of-band verdict
31
+ channel — the inject-tests-after-the-agent-finishes topology) is tracked separate
32
+ work; until then do not treat a passing score from an untrusted adversarial
33
+ candidate as authoritative. The release gate is unaffected: it runs only trusted
34
+ shipped reference code.
35
+ """
36
+
37
+ from __future__ import annotations
38
+
39
+ import json
40
+ import subprocess
41
+ import sys
42
+ import tempfile
43
+ from pathlib import Path
44
+ from typing import Any
45
+
46
+ from ..live._runner import scrubbed_lane_env
47
+
48
+ #: Entry module the candidate code is written to (checks ``import`` this name).
49
+ ENTRY_MODULE = "solution"
50
+ _CHECKS_MODULE = "bench_checks"
51
+ _DEFAULT_TIMEOUT_S = 10.0
52
+
53
+ SUPPORTED_LANGUAGES = ("python",)
54
+
55
+ # The harness-written runner: discovers ``check_*`` callables in the checks
56
+ # module, runs each, and emits a single JSON line of per-check results to stdout.
57
+ # The candidate (``solution.py``) is imported only by the checks module — never
58
+ # by this runner directly — so the oracle stays out of the candidate's reach.
59
+ _PYTHON_RUNNER = """\
60
+ import importlib, json, sys, traceback
61
+ results = {}
62
+ fatal = None
63
+ try:
64
+ checks = importlib.import_module("%(checks)s")
65
+ except Exception:
66
+ fatal = "checks_import_failed: " + traceback.format_exc(limit=2).strip().replace(chr(10), " | ")
67
+ print(json.dumps({"results": {}, "fatal": fatal}))
68
+ sys.exit(1)
69
+ names = sorted(n for n in dir(checks) if n.startswith("check_") and callable(getattr(checks, n)))
70
+ if not names:
71
+ print(json.dumps({"results": {}, "fatal": "no check_* callables found"}))
72
+ sys.exit(1)
73
+ for name in names:
74
+ try:
75
+ getattr(checks, name)()
76
+ results[name] = True
77
+ except Exception:
78
+ results[name] = False
79
+ print(json.dumps({"results": results, "fatal": None}))
80
+ sys.exit(0 if results and all(results.values()) else 1)
81
+ """
82
+
83
+
84
+ def _tail(text: str, limit: int = 2000) -> str:
85
+ text = text or ""
86
+ return text[-limit:]
87
+
88
+
89
+ def _empty_result(explanation: str, raw: dict[str, Any]) -> dict[str, Any]:
90
+ return {
91
+ "result": {
92
+ "scalar": 0.0,
93
+ "components": {"checks_passed": 0.0, "checks_total": 0.0},
94
+ "pass_fail": {},
95
+ "explanation": explanation,
96
+ },
97
+ "raw": raw,
98
+ }
99
+
100
+
101
+ def run_code_tests(
102
+ candidate_code: str,
103
+ checks_code: str,
104
+ *,
105
+ language: str = "python",
106
+ timeout_s: float = _DEFAULT_TIMEOUT_S,
107
+ sandbox: str = "subprocess",
108
+ ) -> dict[str, Any]:
109
+ """Run ``checks_code`` (the held-out oracle) against ``candidate_code``.
110
+
111
+ Returns ``{"result": <unified Result>, "raw": <execution evidence>}``. The
112
+ unified ``Result`` carries a scalar (fraction of checks passed), components
113
+ (passed/total), per-check ``pass_fail`` booleans, and an explanation.
114
+ """
115
+
116
+ if language not in SUPPORTED_LANGUAGES:
117
+ return _empty_result(
118
+ f"unsupported language {language!r}; supported: {SUPPORTED_LANGUAGES}",
119
+ {"sandbox": sandbox, "language": language, "infra_error": True},
120
+ )
121
+ if sandbox == "docker":
122
+ # The Docker lane is opt-in; never silently fall back to a weaker sandbox
123
+ # (that would mislabel isolation).
124
+ from ._docker import run_code_tests_docker # local import: optional lane
125
+
126
+ return run_code_tests_docker(
127
+ candidate_code, checks_code, language=language, timeout_s=timeout_s
128
+ )
129
+ if sandbox != "subprocess":
130
+ return _empty_result(
131
+ f"unknown sandbox {sandbox!r}; expected 'subprocess' or 'docker'",
132
+ {"sandbox": sandbox, "language": language, "infra_error": True},
133
+ )
134
+
135
+ return _run_subprocess(candidate_code, checks_code, timeout_s=timeout_s)
136
+
137
+
138
+ def _run_subprocess(
139
+ candidate_code: str, checks_code: str, *, timeout_s: float
140
+ ) -> dict[str, Any]:
141
+ with tempfile.TemporaryDirectory(prefix="agent-learn-bench-") as tmp:
142
+ root = Path(tmp)
143
+ (root / f"{ENTRY_MODULE}.py").write_text(candidate_code, encoding="utf-8")
144
+ (root / f"{_CHECKS_MODULE}.py").write_text(checks_code, encoding="utf-8")
145
+ (root / "_runner.py").write_text(
146
+ _PYTHON_RUNNER % {"checks": _CHECKS_MODULE}, encoding="utf-8"
147
+ )
148
+
149
+ raw: dict[str, Any] = {
150
+ "sandbox": "subprocess",
151
+ "language": "python",
152
+ "timed_out": False,
153
+ "exit_code": None,
154
+ }
155
+ try:
156
+ proc = subprocess.run(
157
+ [sys.executable, "_runner.py"],
158
+ cwd=str(root),
159
+ env=scrubbed_lane_env(()), # no harness secrets cross into the run
160
+ capture_output=True,
161
+ text=True,
162
+ timeout=timeout_s,
163
+ )
164
+ except subprocess.TimeoutExpired as exc:
165
+ raw["timed_out"] = True
166
+ raw["stdout_tail"] = _tail(exc.stdout if isinstance(exc.stdout, str) else "")
167
+ raw["stderr_tail"] = _tail(exc.stderr if isinstance(exc.stderr, str) else "")
168
+ return _empty_result(f"timed out after {timeout_s}s", raw)
169
+
170
+ raw["exit_code"] = proc.returncode
171
+ raw["stdout_tail"] = _tail(proc.stdout)
172
+ raw["stderr_tail"] = _tail(proc.stderr)
173
+
174
+ parsed = _parse_runner_stdout(proc.stdout)
175
+ if parsed is None:
176
+ return _empty_result(
177
+ f"runner produced no parseable result (exit {proc.returncode})", raw
178
+ )
179
+ fatal = parsed.get("fatal")
180
+ results = {str(k): bool(v) for k, v in (parsed.get("results") or {}).items()}
181
+ if fatal:
182
+ return _empty_result(str(fatal), raw)
183
+ if not results:
184
+ return _empty_result("no checks executed", raw)
185
+
186
+ total = len(results)
187
+ passed = sum(1 for v in results.values() if v)
188
+ return {
189
+ "result": {
190
+ "scalar": round(passed / total, 6),
191
+ "components": {
192
+ "checks_passed": float(passed),
193
+ "checks_total": float(total),
194
+ },
195
+ "pass_fail": results,
196
+ "explanation": f"{passed}/{total} checks passed",
197
+ },
198
+ "raw": raw,
199
+ }
200
+
201
+
202
+ def _parse_runner_stdout(stdout: str) -> dict[str, Any] | None:
203
+ # The runner prints exactly one JSON line; tolerate trailing candidate prints
204
+ # by scanning for the last decodable JSON object.
205
+ for line in reversed((stdout or "").splitlines()):
206
+ line = line.strip()
207
+ if not line.startswith("{"):
208
+ continue
209
+ try:
210
+ return json.loads(line)
211
+ except json.JSONDecodeError:
212
+ continue
213
+ return None
@@ -0,0 +1,215 @@
1
+ """Coding-modality bench suite + the artifact-in runner.
2
+
3
+ A coding suite is a bench-native shape (``agent-learning.bench-suite.v1``): each
4
+ task carries an ``instruction``, a held-out ``checks`` oracle (executed against
5
+ the candidate, never imported by it), a ``reference_solution`` (the gold, used by
6
+ the release gate to prove the verifier accepts a correct answer), and optional
7
+ ``guards``. This is deliberately distinct from the objective-anchored task
8
+ dataset: coding's verdict is "do the held-out tests pass", not a weighted-metric
9
+ mean. The two unify at the Result level, not the suite level — exactly the shape
10
+ the prior-art survey found (task specs are modality-specific; the Task<->Verifier
11
+ coupling and the unified Result are the invariant).
12
+
13
+ ``artifact_in`` control mode scores a *submitted artifact* (candidate code) with
14
+ no live agent — the analogue of patch-scoring harnesses. The agent that produced
15
+ the artifact is out of scope here; only the held-out oracle decides the verdict.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ from pathlib import Path
22
+ from typing import Any, Mapping
23
+
24
+ from ._codeexec import run_code_tests
25
+ from ._grader import GRADING_COMMAND, run_command_graded
26
+
27
+ BENCH_SUITE_KIND = "agent-learning.bench-suite.v1"
28
+
29
+ # A task is graded one of two ways:
30
+ # * "checks" (default, convenience tier): held-out check_* functions import the
31
+ # candidate in-process. Trusted / accidental-gaming only.
32
+ # * "command" (hardened tier): the candidate produces files/output, a held-out
33
+ # grader runs AFTER and emits the verdict via exit code + reward file. Robust
34
+ # against forge + oracle-read; multi-language. See _grader.py.
35
+ _CHECKS_FIELDS = ("id", "instruction", "checks", "reference_solution")
36
+ _COMMAND_FIELDS = ("id", "instruction", "grader_cmd", "grader_files", "reference_files")
37
+
38
+
39
+ class CodingSuiteError(ValueError):
40
+ """Raised for a malformed coding bench suite."""
41
+
42
+
43
+ def is_bench_suite(obj: Any) -> bool:
44
+ return isinstance(obj, Mapping) and obj.get("kind") == BENCH_SUITE_KIND
45
+
46
+
47
+ def _task_grading(task: Mapping[str, Any]) -> str:
48
+ """The grading mode of a task: 'command' (hardened) or 'checks' (convenience)."""
49
+
50
+ mode = task.get("grading")
51
+ if mode in (GRADING_COMMAND, "checks"):
52
+ return str(mode)
53
+ # infer: a grader_cmd ⇒ command-graded; otherwise the legacy checks tier.
54
+ return GRADING_COMMAND if task.get("grader_cmd") else "checks"
55
+
56
+
57
+ def load_coding_suite(obj: Mapping[str, Any] | str | Path) -> dict[str, Any]:
58
+ """Load + validate a coding bench suite (from a path or an in-memory mapping)."""
59
+
60
+ if isinstance(obj, (str, Path)):
61
+ data: Mapping[str, Any] = json.loads(Path(obj).expanduser().read_text("utf-8"))
62
+ else:
63
+ data = obj
64
+ if not is_bench_suite(data):
65
+ raise CodingSuiteError(f"not a {BENCH_SUITE_KIND} suite")
66
+ tasks = data.get("tasks")
67
+ if not isinstance(tasks, list) or not tasks:
68
+ raise CodingSuiteError("coding suite has no tasks")
69
+ seen: set[str] = set()
70
+ for i, task in enumerate(tasks):
71
+ if not isinstance(task, Mapping):
72
+ raise CodingSuiteError(f"task #{i} is not an object")
73
+ required = _COMMAND_FIELDS if _task_grading(task) == GRADING_COMMAND else _CHECKS_FIELDS
74
+ for field in required:
75
+ if not task.get(field):
76
+ raise CodingSuiteError(
77
+ f"task #{i} ({_task_grading(task)}-graded) missing required field {field!r}"
78
+ )
79
+ tid = str(task["id"])
80
+ if tid in seen:
81
+ raise CodingSuiteError(f"duplicate task id {tid!r}")
82
+ seen.add(tid)
83
+ # Every task must declare at least one guard against reward hacking; the
84
+ # held-out oracle is the primary defence, but the suite must say so.
85
+ guards = task.get("guards") or {}
86
+ if int(guards.get("min_guard_count", 0)) < 1:
87
+ raise CodingSuiteError(
88
+ f"task {tid!r} must declare guards.min_guard_count >= 1 "
89
+ "(the held-out-oracle anti-gaming contract)"
90
+ )
91
+ return dict(data)
92
+
93
+
94
+ def _coding_row(
95
+ task: Mapping[str, Any],
96
+ verdict_obj: Mapping[str, Any],
97
+ *,
98
+ evidence_class: str,
99
+ sandbox: str,
100
+ ) -> dict[str, Any]:
101
+ result = dict(verdict_obj["result"])
102
+ scalar = result.get("scalar")
103
+ # All-or-nothing: a coding task is resolved only if EVERY held-out check passes.
104
+ verdict = "pass" if scalar is not None and float(scalar) >= 1.0 else "fail"
105
+ return {
106
+ "task_id": str(task["id"]),
107
+ "modality": "coding",
108
+ "world_kind": "code_exec",
109
+ "control_mode": "artifact_in",
110
+ "result": result,
111
+ "verdict": verdict,
112
+ # The candidate code really executed; honest execution_class is executable.
113
+ "execution_class": "executable",
114
+ "evidence_class": evidence_class,
115
+ # executable + any evidence class is never an overclaim (it really ran).
116
+ "overclaim": False,
117
+ "sandbox": sandbox,
118
+ "raw": verdict_obj.get("raw", {}),
119
+ }
120
+
121
+
122
+ def _void_row(task: Mapping[str, Any], reason: str, *, evidence_class: str) -> dict[str, Any]:
123
+ return {
124
+ "task_id": str(task["id"]),
125
+ "modality": "coding",
126
+ "world_kind": "code_exec",
127
+ "control_mode": "artifact_in",
128
+ "result": {
129
+ "scalar": None,
130
+ "components": {},
131
+ "pass_fail": {},
132
+ "explanation": reason,
133
+ },
134
+ "verdict": "void",
135
+ "execution_class": "executable",
136
+ "evidence_class": evidence_class,
137
+ "overclaim": False,
138
+ "error": reason,
139
+ }
140
+
141
+
142
+ def run_coding_artifact_in(
143
+ suite: Mapping[str, Any],
144
+ submission: Mapping[str, Any],
145
+ *,
146
+ sandbox: str = "subprocess",
147
+ evidence_class: str = "captured_fixture",
148
+ max_tasks: int | None = None,
149
+ default_timeout_s: float = 10.0,
150
+ ) -> list[dict[str, Any]]:
151
+ """Score each task's submitted artifact against its held-out oracle.
152
+
153
+ ``submission`` maps ``task_id -> candidate``. For ``checks``-graded tasks the
154
+ candidate is a source string; for ``command``-graded (hardened) tasks it is a
155
+ ``{path: content}`` file map. A task with no submission is recorded ``void``
156
+ (never silently passed); an infra failure is recorded ``void`` too. Pass the
157
+ :func:`reference_submission` to verify the suite itself (what the gate does).
158
+ """
159
+
160
+ language = str(suite.get("language", "python"))
161
+ # Honesty: a Docker run executes untrusted candidate code under real
162
+ # isolation -> that is a genuine LIVE event, never a fixture/local class.
163
+ # Force at least live_lane (honor an explicit live_stressed); never downgrade.
164
+ if sandbox == "docker" and evidence_class not in ("live_lane", "live_stressed"):
165
+ evidence_class = "live_lane"
166
+ rows: list[dict[str, Any]] = []
167
+ tasks = suite["tasks"]
168
+ if max_tasks is not None:
169
+ tasks = tasks[: max(0, int(max_tasks))]
170
+ for task in tasks:
171
+ tid = str(task["id"])
172
+ candidate = submission.get(tid)
173
+ if candidate is None:
174
+ rows.append(_void_row(task, "no submission provided", evidence_class=evidence_class))
175
+ continue
176
+ timeout_s = float(task.get("timeout_s", default_timeout_s))
177
+ if _task_grading(task) == GRADING_COMMAND:
178
+ files = candidate if isinstance(candidate, Mapping) else {"solution": str(candidate)}
179
+ verdict_obj = run_command_graded(
180
+ task, {str(k): str(v) for k, v in files.items()},
181
+ sandbox=sandbox, timeout_s=timeout_s,
182
+ )
183
+ else:
184
+ verdict_obj = run_code_tests(
185
+ str(candidate), str(task["checks"]),
186
+ language=language, timeout_s=timeout_s, sandbox=sandbox,
187
+ )
188
+ # An infra/config failure (no Docker daemon, image pull failure, bad
189
+ # sandbox/language) means the lane never ran — record it as VOID, never as
190
+ # a real "fail". Conflating "the daemon was missing" with "the agent was
191
+ # wrong" would silently report a correct agent at 0%.
192
+ if (verdict_obj.get("raw") or {}).get("infra_error"):
193
+ reason = (verdict_obj.get("result") or {}).get("explanation") or "infrastructure error"
194
+ rows.append(_void_row(task, f"infra: {reason}", evidence_class=evidence_class))
195
+ continue
196
+ rows.append(
197
+ _coding_row(task, verdict_obj, evidence_class=evidence_class, sandbox=sandbox)
198
+ )
199
+ return rows
200
+
201
+
202
+ def reference_submission(suite: Mapping[str, Any]) -> dict[str, Any]:
203
+ """The gold submission: every task id -> its reference.
204
+
205
+ ``checks``-graded tasks map to a ``reference_solution`` string; ``command``-
206
+ graded tasks map to a ``reference_files`` ``{path: content}`` map.
207
+ """
208
+
209
+ out: dict[str, Any] = {}
210
+ for t in suite["tasks"]:
211
+ if _task_grading(t) == GRADING_COMMAND:
212
+ out[str(t["id"])] = {str(k): str(v) for k, v in (t["reference_files"] or {}).items()}
213
+ else:
214
+ out[str(t["id"])] = str(t["reference_solution"])
215
+ return out
@@ -0,0 +1,237 @@
1
+ """Docker code-exec lane — run held-out checks against untrusted candidate code
2
+ in a per-task, network-isolated, resource-capped, ephemeral container.
3
+
4
+ This is the harder-isolated sibling of the subprocess verifier in ``_codeexec``.
5
+ For **untrusted agent output** it adds real OS-level isolation the subprocess lane
6
+ cannot: no network, no host writes, dropped capabilities + no-new-privileges, a
7
+ nosuid tmpfs, capped CPU/memory/PIDs, killed + removed after the run.
8
+
9
+ Honesty: a Docker run of untrusted candidate code is a genuine **live** event,
10
+ so :func:`fi.alk.bench._coding.run_coding_artifact_in` stamps these rows
11
+ ``evidence_class=live_lane`` (never ``captured_fixture``) — see that module.
12
+
13
+ This lane is **opt-in** (``sandbox="docker"``) and is NEVER a release-gate
14
+ prerequisite: the credential-free ``bench_contract_readiness`` gate runs the
15
+ subprocess lane on trusted shipped code so it works anywhere with no Docker. The
16
+ import is lazy (only resolved when ``sandbox="docker"`` is requested), so the kit
17
+ imports fine on a machine with no Docker.
18
+
19
+ KNOWN LIMITATION (oracle hold-out): hold-out here is only *structural* — the
20
+ checks file is not part of the candidate's source, but it is materialised into the
21
+ same container the candidate runs in, and the candidate's module body executes
22
+ (at import) before the ``check_*`` functions. A deliberately adversarial candidate
23
+ could therefore read the checks file at runtime and reflect the expected values,
24
+ or print a forged result line. This lane defends against *accidental* gaming
25
+ (no-op / fake-success / wrong answer all fail) and gives strong OS isolation, but
26
+ it is NOT yet a hardened defence against a candidate that actively attacks the
27
+ harness protocol. Closing that requires process/UID separation of the oracle from
28
+ the candidate (the inject-tests-only-after-the-agent-finishes topology), tracked
29
+ as the live-agent-in-container step. Do not treat a passing score from an
30
+ untrusted, adversarial candidate as authoritative until that lands.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import base64
36
+ import shutil
37
+ import subprocess
38
+ import uuid
39
+ from typing import Any
40
+
41
+ from ._codeexec import (
42
+ SUPPORTED_LANGUAGES,
43
+ _empty_result,
44
+ _parse_runner_stdout,
45
+ _tail,
46
+ )
47
+
48
+ # In-container bootstrap: materialise the candidate + held-out checks from base64
49
+ # into the writable tmpfs (no host bind-mount — works identically on macOS Docker
50
+ # Desktop and Linux), then run each check_* in isolation and emit one JSON line.
51
+ # Doubled braces are literal dict syntax preserved through ``str.format``.
52
+ _DOCKER_BOOTSTRAP = (
53
+ "import base64,importlib,json,sys,traceback\n"
54
+ "open('/tmp/solution.py','wb').write(base64.b64decode('{cand_b64}'))\n"
55
+ "open('/tmp/bench_checks.py','wb').write(base64.b64decode('{checks_b64}'))\n"
56
+ "sys.path.insert(0,'/tmp')\n"
57
+ "results={{}}\n"
58
+ "try:\n"
59
+ " checks=importlib.import_module('bench_checks')\n"
60
+ "except Exception:\n"
61
+ " print(json.dumps({{'results':{{}},'fatal':'checks_import_failed: '"
62
+ "+traceback.format_exc(limit=2).strip().replace(chr(10),' | ')}}));sys.exit(1)\n"
63
+ "names=sorted(n for n in dir(checks) if n.startswith('check_') and callable(getattr(checks,n)))\n"
64
+ "if not names:\n"
65
+ " print(json.dumps({{'results':{{}},'fatal':'no check_* callables found'}}));sys.exit(1)\n"
66
+ "for name in names:\n"
67
+ " try:\n"
68
+ " getattr(checks,name)();results[name]=True\n"
69
+ " except Exception:\n"
70
+ " results[name]=False\n"
71
+ "print(json.dumps({{'results':results,'fatal':None}}))\n"
72
+ "sys.exit(0 if results and all(results.values()) else 1)\n"
73
+ )
74
+
75
+ # Default base image. Production should pin by digest (image@sha256:...) for
76
+ # determinism; the tag default keeps the example/proof portable.
77
+ DEFAULT_IMAGE = "python:3.11-slim"
78
+
79
+ _DEFAULT_MEMORY = "256m"
80
+ _DEFAULT_CPUS = "1.0"
81
+ _DEFAULT_PIDS = 128
82
+
83
+
84
+ def docker_available() -> bool:
85
+ """True if a working Docker daemon is reachable (cheap, no pull)."""
86
+
87
+ if shutil.which("docker") is None:
88
+ return False
89
+ try:
90
+ proc = subprocess.run(
91
+ ["docker", "info", "--format", "{{.ServerVersion}}"],
92
+ capture_output=True,
93
+ text=True,
94
+ timeout=15,
95
+ )
96
+ except Exception:
97
+ return False
98
+ return proc.returncode == 0
99
+
100
+
101
+ def _build_docker_argv(
102
+ name: str, image: str, memory: str, cpus: str, bootstrap: str
103
+ ) -> list[str]:
104
+ """Build the hardened ``docker run`` argv (pure; unit-testable without a daemon).
105
+
106
+ Defense-in-depth for the untrusted lane: no network, read-only rootfs, a
107
+ non-root user, ALL capabilities dropped (the bounding set too — uid 65534 only
108
+ clears effective/permitted, but the base image ships setuid-root binaries that
109
+ could otherwise re-escalate), no new privileges, a nosuid size-capped tmpfs as
110
+ the only writable surface, and PID/memory/CPU caps. Per-task, ephemeral
111
+ (``--rm``); args passed as a list (never a shell).
112
+ """
113
+
114
+ return [
115
+ "docker", "run", "--rm",
116
+ "--name", name,
117
+ "--network", "none", # the real win the subprocess lane can't give
118
+ "--memory", memory,
119
+ "--cpus", cpus,
120
+ "--pids-limit", str(_DEFAULT_PIDS),
121
+ "--user", "65534:65534", # nobody: no in-container root
122
+ "--cap-drop", "ALL", # drop the bounding set (block setuid re-escalation)
123
+ "--security-opt", "no-new-privileges",
124
+ "--read-only", # rootfs read-only; only the tmpfs is writable
125
+ "--tmpfs", "/tmp:size=16m,nosuid", # nosec B108 — container-internal path, not a host temp
126
+ "--entrypoint", "python",
127
+ image,
128
+ "-B", "-c", bootstrap, # -B: no .pyc writes under the read-only fs
129
+ ]
130
+
131
+
132
+ def run_code_tests_docker(
133
+ candidate_code: str,
134
+ checks_code: str,
135
+ *,
136
+ language: str = "python",
137
+ timeout_s: float = 10.0,
138
+ image: str = DEFAULT_IMAGE,
139
+ memory: str = _DEFAULT_MEMORY,
140
+ cpus: str = _DEFAULT_CPUS,
141
+ ) -> dict[str, Any]:
142
+ """Run ``checks_code`` against ``candidate_code`` inside an isolated container.
143
+
144
+ Returns the same ``{"result", "raw"}`` shape as the subprocess verifier; the
145
+ ``raw`` block records ``sandbox="docker"``, the image, and isolation flags.
146
+ Never raises for an unavailable daemon or a hostile candidate — both surface
147
+ as an honest failing Result.
148
+ """
149
+
150
+ if language not in SUPPORTED_LANGUAGES:
151
+ return _empty_result(
152
+ f"unsupported language {language!r}; supported: {SUPPORTED_LANGUAGES}",
153
+ {"sandbox": "docker", "language": language, "infra_error": True},
154
+ )
155
+ if not docker_available():
156
+ return _empty_result(
157
+ "docker unavailable (no daemon / not installed)",
158
+ {"sandbox": "docker", "language": language,
159
+ "docker_available": False, "infra_error": True},
160
+ )
161
+
162
+ name = f"agent-learn-bench-{uuid.uuid4().hex[:12]}"
163
+ raw: dict[str, Any] = {
164
+ "sandbox": "docker",
165
+ "language": language,
166
+ "image": image,
167
+ "network": "none",
168
+ "memory": memory,
169
+ "cpus": cpus,
170
+ "cap_drop": "all",
171
+ "no_new_privileges": True,
172
+ "container": name,
173
+ "timed_out": False,
174
+ "exit_code": None,
175
+ }
176
+
177
+ bootstrap = _DOCKER_BOOTSTRAP.format(
178
+ cand_b64=base64.b64encode(candidate_code.encode("utf-8")).decode("ascii"),
179
+ checks_b64=base64.b64encode(checks_code.encode("utf-8")).decode("ascii"),
180
+ )
181
+ argv = _build_docker_argv(name, image, memory, cpus, bootstrap)
182
+ try:
183
+ proc = subprocess.run(
184
+ argv, capture_output=True, text=True, timeout=timeout_s + 20.0
185
+ )
186
+ except subprocess.TimeoutExpired as exc:
187
+ raw["timed_out"] = True
188
+ raw["stdout_tail"] = _tail(exc.stdout if isinstance(exc.stdout, str) else "")
189
+ raw["stderr_tail"] = _tail(exc.stderr if isinstance(exc.stderr, str) else "")
190
+ _force_kill(name)
191
+ return _empty_result(f"timed out after {timeout_s}s (container killed)", raw)
192
+
193
+ raw["exit_code"] = proc.returncode
194
+ raw["stdout_tail"] = _tail(proc.stdout)
195
+ raw["stderr_tail"] = _tail(proc.stderr)
196
+
197
+ # A daemon/image error (e.g. image not pulled) is infra, not an agent fail.
198
+ if proc.returncode not in (0, 1) and "{" not in (proc.stdout or ""):
199
+ raw["infra_error"] = True
200
+ return _empty_result(
201
+ f"docker run failed (exit {proc.returncode}): {_tail(proc.stderr, 300)}",
202
+ raw,
203
+ )
204
+
205
+ parsed = _parse_runner_stdout(proc.stdout)
206
+ if parsed is None:
207
+ return _empty_result(
208
+ f"runner produced no parseable result (exit {proc.returncode})", raw
209
+ )
210
+ fatal = parsed.get("fatal")
211
+ results = {str(k): bool(v) for k, v in (parsed.get("results") or {}).items()}
212
+ if fatal:
213
+ return _empty_result(str(fatal), raw)
214
+ if not results:
215
+ return _empty_result("no checks executed", raw)
216
+
217
+ total = len(results)
218
+ passed = sum(1 for v in results.values() if v)
219
+ return {
220
+ "result": {
221
+ "scalar": round(passed / total, 6),
222
+ "components": {
223
+ "checks_passed": float(passed),
224
+ "checks_total": float(total),
225
+ },
226
+ "pass_fail": results,
227
+ "explanation": f"{passed}/{total} checks passed",
228
+ },
229
+ "raw": raw,
230
+ }
231
+
232
+
233
+ def _force_kill(name: str) -> None:
234
+ try:
235
+ subprocess.run(["docker", "kill", name], capture_output=True, timeout=15)
236
+ except Exception: # nosec B110 — best-effort cleanup; a kill failure must never propagate
237
+ pass