agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,286 @@
1
+ """Command/artifact-graded coding lane — the hardened coding tier.
2
+
3
+ This resolves the in-process forge/oracle-read weakness of the ``check_*`` lane by
4
+ changing the *model*, not bolting on isolation. A task gives the candidate a
5
+ working directory + a way to RUN it; a **held-out grader** runs AFTERWARD and
6
+ emits the verdict via its **exit code + a reward file in a grader-controlled
7
+ path** — never parsed from candidate-shared stdout.
8
+
9
+ Why this is robust (the two vulns from the PR review, structurally closed):
10
+
11
+ * **No verdict forgery** — the verdict is the grader's exit code (and an optional
12
+ ``reward.json`` the grader writes), not anything the candidate prints.
13
+ * **No oracle read** — the grader files (held-out expected values / tests) are
14
+ written ONLY after the candidate command has finished and its processes are
15
+ killed. The candidate never co-runs with the grader, so it cannot read the
16
+ expected values; and in the Docker lane the grader files are owned by a
17
+ different user the candidate uid cannot read.
18
+
19
+ It also gives **multi-language for free**: the candidate ``build`` command and the
20
+ ``grader`` command are arbitrary shell, so the same lane grades Python, bash,
21
+ Node, compiled languages, etc.
22
+
23
+ Two sandboxes share one flow (temporal separation candidate→grader):
24
+ * ``subprocess`` — candidate + grader run as host subprocesses in *separate*
25
+ temp dirs; the grader dir path is given only to the grader. Credential-free,
26
+ Docker-free; used by the release gate on trusted shipped tasks.
27
+ * ``docker`` — a per-task, network-off, capped, ephemeral container; candidate
28
+ runs as an unprivileged uid, grader files land in a root-owned dir the
29
+ candidate cannot read. The hardened lane for untrusted agent output.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import json
35
+ import os
36
+ import subprocess
37
+ import tempfile
38
+ import uuid
39
+ from pathlib import Path
40
+ from typing import Any, Mapping
41
+
42
+ from ._codeexec import _empty_result, _tail
43
+
44
+ GRADING_COMMAND = "command" # suite/task grading mode discriminator
45
+ _DEFAULT_TIMEOUT_S = 20.0
46
+ _REWARD_FILE = "reward.json"
47
+
48
+
49
+ def _files(value: Any) -> dict[str, str]:
50
+ """Coerce a {path: content} mapping to str->str (defensive)."""
51
+
52
+ if not isinstance(value, Mapping):
53
+ return {}
54
+ return {str(k): str(v) for k, v in value.items()}
55
+
56
+
57
+ def _write_tree(root: Path, files: Mapping[str, str]) -> None:
58
+ for rel, content in files.items():
59
+ dest = root / rel
60
+ dest.parent.mkdir(parents=True, exist_ok=True)
61
+ dest.write_text(content, encoding="utf-8")
62
+
63
+
64
+ def _result_from_grader(
65
+ rc: int | None, reward: Mapping[str, Any] | None, raw: dict[str, Any]
66
+ ) -> dict[str, Any]:
67
+ """Build the unified Result from the grader's exit code + optional reward.json.
68
+
69
+ Verdict authority: the grader's exit code (0 = pass). If the grader also wrote
70
+ a ``reward.json`` with a numeric ``score`` in [0,1], that becomes the scalar;
71
+ otherwise the scalar is 1.0/0.0 from the exit code. Sub-check booleans, if the
72
+ grader reports a ``checks`` map, flow into ``pass_fail``.
73
+ """
74
+
75
+ passed = rc == 0
76
+ scalar: float
77
+ pass_fail: dict[str, bool] = {}
78
+ explanation = f"grader exit {rc}"
79
+ if isinstance(reward, Mapping):
80
+ score = reward.get("score")
81
+ if isinstance(score, (int, float)) and not isinstance(score, bool):
82
+ scalar = round(float(score), 6)
83
+ else:
84
+ scalar = 1.0 if passed else 0.0
85
+ checks = reward.get("checks")
86
+ if isinstance(checks, Mapping):
87
+ pass_fail = {str(k): bool(v) for k, v in checks.items()}
88
+ if reward.get("explanation"):
89
+ explanation = str(reward["explanation"])
90
+ else:
91
+ scalar = 1.0 if passed else 0.0
92
+ if not pass_fail:
93
+ pass_fail = {"grader": passed}
94
+ return {
95
+ "result": {
96
+ "scalar": scalar,
97
+ "components": {"grader_exit_ok": 1.0 if passed else 0.0},
98
+ "pass_fail": pass_fail,
99
+ "explanation": explanation,
100
+ },
101
+ "raw": raw,
102
+ }
103
+
104
+
105
+ def run_command_graded(
106
+ task: Mapping[str, Any],
107
+ candidate_files: Mapping[str, str],
108
+ *,
109
+ sandbox: str = "subprocess",
110
+ timeout_s: float = _DEFAULT_TIMEOUT_S,
111
+ ) -> dict[str, Any]:
112
+ """Grade ``candidate_files`` for a command-graded ``task``.
113
+
114
+ Returns ``{"result", "raw"}``. Never raises for infra problems (missing Docker,
115
+ grader crash) — those surface as a failing/infra Result, tagged
116
+ ``raw["infra_error"]`` when the lane could not run at all.
117
+ """
118
+
119
+ if sandbox == "docker":
120
+ from ._docker import docker_available
121
+
122
+ if not docker_available():
123
+ return _empty_result(
124
+ "docker unavailable (no daemon / not installed)",
125
+ {"sandbox": "docker", "grading": GRADING_COMMAND, "infra_error": True},
126
+ )
127
+ return _run_docker_graded(task, candidate_files, timeout_s=timeout_s)
128
+ if sandbox != "subprocess":
129
+ return _empty_result(
130
+ f"unknown sandbox {sandbox!r}; expected 'subprocess' or 'docker'",
131
+ {"sandbox": sandbox, "grading": GRADING_COMMAND, "infra_error": True},
132
+ )
133
+ return _run_subprocess_graded(task, candidate_files, timeout_s=timeout_s)
134
+
135
+
136
+ def _run_subprocess_graded(
137
+ task: Mapping[str, Any], candidate_files: Mapping[str, str], *, timeout_s: float
138
+ ) -> dict[str, Any]:
139
+ build = task.get("build")
140
+ grader_cmd = str(task.get("grader_cmd") or "")
141
+ if not grader_cmd:
142
+ return _empty_result("task has no grader_cmd", {"sandbox": "subprocess", "infra_error": True})
143
+
144
+ raw: dict[str, Any] = {"sandbox": "subprocess", "grading": GRADING_COMMAND, "timed_out": False}
145
+ with tempfile.TemporaryDirectory(prefix="bench-work-") as work_s, \
146
+ tempfile.TemporaryDirectory(prefix="bench-grader-") as grader_s:
147
+ work = Path(work_s)
148
+ grader = Path(grader_s)
149
+ _write_tree(work, {**_files(task.get("files")), **dict(candidate_files)})
150
+
151
+ # PHASE 1 — candidate runs with NO grader present (temporal hold-out).
152
+ if build:
153
+ try:
154
+ proc = subprocess.run(
155
+ ["sh", "-c", str(build)], cwd=str(work), capture_output=True,
156
+ text=True, timeout=timeout_s,
157
+ )
158
+ raw["build_exit"] = proc.returncode
159
+ raw["build_stdout_tail"] = _tail(proc.stdout)
160
+ except subprocess.TimeoutExpired:
161
+ raw["timed_out"] = True
162
+ return _empty_result(f"candidate build timed out after {timeout_s}s", raw)
163
+
164
+ # PHASE 2 — grader written AFTER, in a dir the candidate phase never knew.
165
+ _write_tree(grader, _files(task.get("grader_files")))
166
+ # The subprocess lane is the trusted/gate tier (not a security boundary —
167
+ # the Docker lane is), so the grader gets a usable PATH to resolve
168
+ # interpreters; GRADER_DIR points it at its held-out files.
169
+ env = {
170
+ "PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"),
171
+ "GRADER_DIR": str(grader),
172
+ "HOME": os.environ.get("HOME", str(work)),
173
+ }
174
+ try:
175
+ gproc = subprocess.run(
176
+ ["sh", "-c", grader_cmd], cwd=str(work), capture_output=True,
177
+ text=True, timeout=timeout_s, env=env,
178
+ )
179
+ except subprocess.TimeoutExpired:
180
+ raw["timed_out"] = True
181
+ return _empty_result(f"grader timed out after {timeout_s}s", raw)
182
+ raw["grader_exit"] = gproc.returncode
183
+ raw["grader_stdout_tail"] = _tail(gproc.stdout)
184
+ raw["grader_stderr_tail"] = _tail(gproc.stderr)
185
+
186
+ reward = _read_reward(grader / _REWARD_FILE)
187
+ return _result_from_grader(gproc.returncode, reward, raw)
188
+
189
+
190
+ def _read_reward(path: Path) -> dict[str, Any] | None:
191
+ try:
192
+ if path.exists():
193
+ return json.loads(path.read_text(encoding="utf-8"))
194
+ except (OSError, json.JSONDecodeError):
195
+ return None
196
+ return None
197
+
198
+
199
+ # ---- Docker command-graded lane (hardened, opt-in) ----
200
+
201
+ _DOCKER_IMAGE = "python:3.11-slim"
202
+
203
+
204
+ def _docker(*args: str, timeout: float = 30.0) -> subprocess.CompletedProcess:
205
+ return subprocess.run(
206
+ ["docker", *args], capture_output=True, text=True, timeout=timeout
207
+ )
208
+
209
+
210
+ def _run_docker_graded(
211
+ task: Mapping[str, Any], candidate_files: Mapping[str, str], *, timeout_s: float
212
+ ) -> dict[str, Any]:
213
+ build = task.get("build")
214
+ grader_cmd = str(task.get("grader_cmd") or "")
215
+ image = str(task.get("image") or _DOCKER_IMAGE)
216
+ if not grader_cmd:
217
+ return _empty_result("task has no grader_cmd", {"sandbox": "docker", "infra_error": True})
218
+
219
+ name = f"agent-learn-grade-{uuid.uuid4().hex[:12]}"
220
+ raw: dict[str, Any] = {
221
+ "sandbox": "docker", "grading": GRADING_COMMAND, "image": image,
222
+ "network": "none", "container": name, "timed_out": False,
223
+ }
224
+ with tempfile.TemporaryDirectory(prefix="bench-docker-") as host_s:
225
+ host = Path(host_s)
226
+ work_host = host / "work"
227
+ grader_host = host / "grader"
228
+ _write_tree(work_host, {**_files(task.get("files")), **dict(candidate_files)})
229
+ _write_tree(grader_host, _files(task.get("grader_files")))
230
+
231
+ started = _docker(
232
+ "run", "-d", "--rm", "--name", name, "--network", "none",
233
+ "--memory", "512m", "--cpus", "1.0", "--pids-limit", "256",
234
+ "--cap-drop", "ALL", "--security-opt", "no-new-privileges",
235
+ image, "sleep", str(int(timeout_s * 2 + 60)),
236
+ )
237
+ if started.returncode != 0:
238
+ raw["infra_error"] = True
239
+ return _empty_result(
240
+ f"docker run failed: {_tail(started.stderr, 300)}", raw
241
+ )
242
+ try:
243
+ # candidate user + dirs; /work candidate-writable, /grader root-only.
244
+ _docker("exec", name, "sh", "-c",
245
+ "id cand 2>/dev/null || useradd -M -s /usr/sbin/nologin cand; "
246
+ "mkdir -p /work /grader")
247
+ _docker("cp", f"{work_host}/.", f"{name}:/work")
248
+ _docker("exec", name, "sh", "-c", "chown -R cand:cand /work && chmod 700 /grader")
249
+
250
+ # PHASE 1 — candidate runs as `cand`, no grader files present yet.
251
+ if build:
252
+ try:
253
+ b = _docker("exec", "-u", "cand", "-w", "/work", name,
254
+ "sh", "-c", str(build), timeout=timeout_s + 15)
255
+ raw["build_exit"] = b.returncode
256
+ raw["build_stdout_tail"] = _tail(b.stdout)
257
+ except subprocess.TimeoutExpired:
258
+ raw["timed_out"] = True
259
+ return _empty_result(f"candidate build timed out after {timeout_s}s", raw)
260
+
261
+ # kill any lingering candidate processes before grading.
262
+ _docker("exec", name, "sh", "-c", "pkill -u cand 2>/dev/null || true")
263
+
264
+ # PHASE 2 — inject grader (root-owned, unreadable to cand), run as root.
265
+ _docker("cp", f"{grader_host}/.", f"{name}:/grader")
266
+ _docker("exec", name, "sh", "-c", "chown -R root:root /grader && chmod -R go-rwx /grader")
267
+ try:
268
+ g = _docker("exec", "-w", "/work", name, "sh", "-c",
269
+ f"export GRADER_DIR=/grader; {grader_cmd}", timeout=timeout_s + 15)
270
+ except subprocess.TimeoutExpired:
271
+ raw["timed_out"] = True
272
+ return _empty_result(f"grader timed out after {timeout_s}s", raw)
273
+ raw["grader_exit"] = g.returncode
274
+ raw["grader_stdout_tail"] = _tail(g.stdout)
275
+ raw["grader_stderr_tail"] = _tail(g.stderr)
276
+
277
+ cat = _docker("exec", name, "sh", "-c", "cat /grader/reward.json 2>/dev/null || true")
278
+ reward: dict[str, Any] | None = None
279
+ if cat.stdout.strip():
280
+ try:
281
+ reward = json.loads(cat.stdout)
282
+ except json.JSONDecodeError:
283
+ reward = None
284
+ return _result_from_grader(g.returncode, reward, raw)
285
+ finally:
286
+ _docker("rm", "-f", name)
fi/alk/bench/_pull.py ADDED
@@ -0,0 +1,212 @@
1
+ """Pull / RL control mode — the AGENT drives a live environment via reset/step.
2
+
3
+ The push lane has the harness drive the agent; artifact-in scores a submitted
4
+ artifact. **Pull** inverts control: the agent is a policy ``obs -> action`` that
5
+ steps an environment until done, and the score is the environment's reward. This
6
+ is the Gym/OpenEnv shape, run live (not replayed).
7
+
8
+ Deep-contract + simulated: the environments here are deterministic, in-process,
9
+ credential-free simulators (so the lane is fully gate-verifiable). A *live*
10
+ external env server (an HTTP step/reset endpoint) is the same contract with a
11
+ network transport and is deferred to owner infra — it plugs in as another
12
+ ``Environment`` without changing the driver or the unified Result.
13
+
14
+ An environment implements:
15
+ * ``reset(spec) -> (state, obs)``
16
+ * ``step(state, action) -> (state, obs, reward, done, info)``
17
+ * ``optimal_action(obs) -> action`` — a reference policy (proves solvability)
18
+ * ``actions`` — the discrete action set
19
+
20
+ A policy is a callable ``obs -> action`` or a spec dict: ``{"type": "reference"}``
21
+ (the env's optimal policy) or ``{"type": "noop"}`` (always the first action).
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from typing import Any, Callable, Mapping, Protocol
27
+
28
+
29
+ class Environment(Protocol):
30
+ actions: tuple[str, ...]
31
+
32
+ def reset(self, spec: Mapping[str, Any]) -> tuple[dict, dict]: ...
33
+ def step(self, state: dict, action: str) -> tuple[dict, dict, float, bool, dict]: ...
34
+ def optimal_action(self, obs: Mapping[str, Any]) -> str: ...
35
+
36
+
37
+ class ReachTargetEnv:
38
+ """1-D navigation: move toward ``target`` from ``start`` within ``max_steps``.
39
+
40
+ obs = {pos, target, remaining}. Reward 1.0 the step the agent lands on target
41
+ (then done); 0.0 otherwise. Deterministic + trivially verifiable; the optimal
42
+ policy is "step toward target".
43
+ """
44
+
45
+ actions: tuple[str, ...] = ("left", "right", "stay")
46
+
47
+ def reset(self, spec: Mapping[str, Any]) -> tuple[dict, dict]:
48
+ state = {
49
+ "pos": int(spec.get("start", 0)),
50
+ "target": int(spec.get("target", 5)),
51
+ "steps": 0,
52
+ "max_steps": int(spec.get("max_steps", 20)),
53
+ }
54
+ return state, self._obs(state)
55
+
56
+ def step(self, state: dict, action: str) -> tuple[dict, dict, float, bool, dict]:
57
+ state = dict(state)
58
+ state["pos"] += {"left": -1, "right": 1, "stay": 0}.get(action, 0)
59
+ state["steps"] += 1
60
+ reached = state["pos"] == state["target"]
61
+ done = reached or state["steps"] >= state["max_steps"]
62
+ reward = 1.0 if reached else 0.0
63
+ return state, self._obs(state), reward, done, {"reached": reached}
64
+
65
+ def optimal_action(self, obs: Mapping[str, Any]) -> str:
66
+ if obs["pos"] < obs["target"]:
67
+ return "right"
68
+ if obs["pos"] > obs["target"]:
69
+ return "left"
70
+ return "stay"
71
+
72
+ @staticmethod
73
+ def _obs(state: Mapping[str, Any]) -> dict:
74
+ return {
75
+ "pos": state["pos"],
76
+ "target": state["target"],
77
+ "remaining": state["max_steps"] - state["steps"],
78
+ }
79
+
80
+
81
+ class GuessNumberEnv:
82
+ """Binary-search style: guess ``secret`` in [low, high] with higher/lower hints.
83
+
84
+ obs = {low, high, last, hint, remaining}. Reward 1.0 on the correct guess.
85
+ Optimal policy = guess the midpoint. Action = the integer guess (as str).
86
+ """
87
+
88
+ actions: tuple[str, ...] = () # any int in range; reference uses midpoint
89
+
90
+ def reset(self, spec: Mapping[str, Any]) -> tuple[dict, dict]:
91
+ low, high = int(spec.get("low", 1)), int(spec.get("high", 100))
92
+ state = {
93
+ "low": low, "high": high, "secret": int(spec.get("secret", (low + high) // 3)),
94
+ "last": None, "hint": "go", "steps": 0,
95
+ "max_steps": int(spec.get("max_steps", 12)),
96
+ }
97
+ return state, self._obs(state)
98
+
99
+ def step(self, state: dict, action: str) -> tuple[dict, dict, float, bool, dict]:
100
+ state = dict(state)
101
+ try:
102
+ guess = int(action)
103
+ except (TypeError, ValueError):
104
+ guess = state["low"]
105
+ state["steps"] += 1
106
+ state["last"] = guess
107
+ if guess == state["secret"]:
108
+ state["hint"] = "correct"
109
+ return state, self._obs(state), 1.0, True, {"reached": True}
110
+ if guess < state["secret"]:
111
+ state["low"] = guess + 1
112
+ state["hint"] = "higher"
113
+ else:
114
+ state["high"] = guess - 1
115
+ state["hint"] = "lower"
116
+ done = state["steps"] >= state["max_steps"]
117
+ return state, self._obs(state), 0.0, done, {"reached": False}
118
+
119
+ def optimal_action(self, obs: Mapping[str, Any]) -> str:
120
+ return str((int(obs["low"]) + int(obs["high"])) // 2)
121
+
122
+ @staticmethod
123
+ def _obs(state: Mapping[str, Any]) -> dict:
124
+ return {
125
+ "low": state["low"], "high": state["high"], "last": state["last"],
126
+ "hint": state["hint"], "remaining": state["max_steps"] - state["steps"],
127
+ }
128
+
129
+
130
+ ENVIRONMENTS: dict[str, Callable[[], Environment]] = {
131
+ "reach_target": lambda: ReachTargetEnv(),
132
+ "guess_number": lambda: GuessNumberEnv(),
133
+ }
134
+
135
+
136
+ class PullError(ValueError):
137
+ """Raised for an unknown env kind or malformed pull task."""
138
+
139
+
140
+ def resolve_policy(agent: Any, env: Environment) -> Callable[[Mapping[str, Any]], str]:
141
+ """Resolve a policy from a callable or a spec dict (``reference`` / ``noop``)."""
142
+
143
+ if callable(agent):
144
+ return agent
145
+ spec = agent if isinstance(agent, Mapping) else {}
146
+ kind = str(spec.get("type", "reference"))
147
+ if kind == "reference":
148
+ return env.optimal_action
149
+ if kind == "noop":
150
+ first = env.actions[0] if env.actions else "0"
151
+ return lambda _obs: first
152
+ raise PullError(f"unknown pull policy {kind!r}; expected callable / reference / noop")
153
+
154
+
155
+ def run_pull(task: Mapping[str, Any], agent: Any) -> dict[str, Any]:
156
+ """Run one agent-driven episode over a simulated environment.
157
+
158
+ Returns ``{"result", "raw"}`` (unified Result). The scalar is the cumulative
159
+ reward; ``pass_fail`` records goal-reached; ``raw`` records the trajectory
160
+ length + terminal info.
161
+ """
162
+
163
+ env_spec = task.get("env") or {}
164
+ kind = str(env_spec.get("kind") or "")
165
+ if kind not in ENVIRONMENTS:
166
+ return {
167
+ "result": {"scalar": 0.0, "components": {}, "pass_fail": {},
168
+ "explanation": f"unknown env kind {kind!r}"},
169
+ "raw": {"control": "pull", "infra_error": True, "env_kind": kind},
170
+ }
171
+ env = ENVIRONMENTS[kind]()
172
+ try:
173
+ policy = resolve_policy(agent, env)
174
+ except PullError as exc:
175
+ return {
176
+ "result": {"scalar": 0.0, "components": {}, "pass_fail": {},
177
+ "explanation": str(exc)},
178
+ "raw": {"control": "pull", "infra_error": True},
179
+ }
180
+
181
+ state, obs = env.reset(env_spec.get("spec") or {})
182
+ total = 0.0
183
+ reached = False
184
+ steps = 0
185
+ hard_cap = int((env_spec.get("spec") or {}).get("max_steps", 50)) + 5
186
+ while steps < hard_cap:
187
+ try:
188
+ action = policy(obs)
189
+ except Exception as exc: # a misbehaving policy fails the episode, not the lane
190
+ return {
191
+ "result": {"scalar": round(total, 6), "components": {"reward": total},
192
+ "pass_fail": {"goal_reached": False},
193
+ "explanation": f"policy raised: {exc}"},
194
+ "raw": {"control": "pull", "env_kind": kind, "steps": steps, "policy_error": True},
195
+ }
196
+ state, obs, reward, done, info = env.step(state, str(action))
197
+ total += float(reward)
198
+ steps += 1
199
+ if info.get("reached"):
200
+ reached = True
201
+ if done:
202
+ break
203
+
204
+ return {
205
+ "result": {
206
+ "scalar": round(total, 6),
207
+ "components": {"reward": round(total, 6), "steps": float(steps)},
208
+ "pass_fail": {"goal_reached": reached},
209
+ "explanation": f"{'reached' if reached else 'did not reach'} goal in {steps} steps",
210
+ },
211
+ "raw": {"control": "pull", "env_kind": kind, "steps": steps, "reached": reached},
212
+ }
fi/alk/bench/_voice.py ADDED
@@ -0,0 +1,147 @@
1
+ """Voice modality — deterministic voice-episode verifier (deep contract + simulated).
2
+
3
+ Voice is the modality that stress-tests the harness: the environment is an active
4
+ caller and the verifier is *temporal*, not an exit code. This module scores a
5
+ voice **episode transcript** (interleaved caller + agent turns with millisecond
6
+ timing) on the dimensions a real voice benchmark cares about:
7
+
8
+ * **latency** — the agent answers within the budget after the caller stops;
9
+ * **turn-taking** — no harmful overlap (both speaking at once) outside a
10
+ legitimate barge-in;
11
+ * **barge-in handling** — when the caller interrupts mid-agent-turn, the agent
12
+ yields promptly;
13
+ * **task content** — the agent's words cover the required content.
14
+
15
+ This is the **simulated / deep-contract** tier: it scores a transcript produced
16
+ by a deterministic simulated caller, so it is fully credential-free and
17
+ gate-verifiable. The same verifier consumes a transcript captured from a *live*
18
+ audio/SIP/WebRTC call + ASR — that live capture (and real WER) is deferred to
19
+ owner infra; it plugs in here unchanged by producing the same transcript shape.
20
+
21
+ A transcript is a list of turns::
22
+
23
+ {"speaker": "caller"|"agent", "start_ms": int, "end_ms": int,
24
+ "text": str, "interrupt": bool (optional, caller turns only)}
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ from typing import Any, Mapping, Sequence
30
+
31
+ _DEFAULT_MAX_LATENCY_MS = 1200
32
+ _BARGE_IN_YIELD_MS = 600 # the agent must stop within this of a barge-in to "yield"
33
+ _PASS_FLOOR = 0.75 # each sub-score must meet this for a pass
34
+
35
+
36
+ def _norm_turns(dialogue: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
37
+ turns = []
38
+ for t in dialogue:
39
+ turns.append({
40
+ "speaker": str(t.get("speaker")),
41
+ "start_ms": int(t.get("start_ms", 0)),
42
+ "end_ms": int(t.get("end_ms", 0)),
43
+ "text": str(t.get("text") or ""),
44
+ "interrupt": bool(t.get("interrupt", False)),
45
+ })
46
+ return turns
47
+
48
+
49
+ def _latency_score(turns: list[dict[str, Any]], max_latency_ms: int) -> float:
50
+ gaps_ok, gaps = 0, 0
51
+ for i, t in enumerate(turns):
52
+ if t["speaker"] != "agent":
53
+ continue
54
+ prev = next((turns[j] for j in range(i - 1, -1, -1)
55
+ if turns[j]["speaker"] == "caller"), None)
56
+ if prev is None:
57
+ continue
58
+ gap = t["start_ms"] - prev["end_ms"]
59
+ gaps += 1
60
+ if 0 <= gap <= max_latency_ms:
61
+ gaps_ok += 1
62
+ return gaps_ok / gaps if gaps else 1.0
63
+
64
+
65
+ def _overlap_and_bargein(turns: list[dict[str, Any]]) -> tuple[float, float]:
66
+ """Turn-taking (no harmful overlap) + barge-in handling scores."""
67
+
68
+ agent_turns = [t for t in turns if t["speaker"] == "agent"]
69
+ harmful, considered = 0, 0
70
+ bargein_handled, bargein_total = 0, 0
71
+ callers = [t for t in turns if t["speaker"] == "caller"]
72
+ for a in agent_turns:
73
+ considered += 1
74
+ overlapping_callers = [
75
+ c for c in callers
76
+ if c["start_ms"] < a["end_ms"] and c["end_ms"] > a["start_ms"]
77
+ ]
78
+ legit_bargein = False
79
+ for c in overlapping_callers:
80
+ if c["interrupt"]:
81
+ bargein_total += 1
82
+ legit_bargein = True
83
+ # the agent must yield: its turn ends within the window after the
84
+ # interrupt begins.
85
+ if a["end_ms"] - c["start_ms"] <= _BARGE_IN_YIELD_MS:
86
+ bargein_handled += 1
87
+ if overlapping_callers and not legit_bargein:
88
+ harmful += 1 # both speaking at once with no barge-in to excuse it
89
+ turn_taking = 1.0 - (harmful / considered) if considered else 1.0
90
+ bargein = bargein_handled / bargein_total if bargein_total else 1.0
91
+ return turn_taking, bargein
92
+
93
+
94
+ def _content_score(turns: list[dict[str, Any]], required: Sequence[str]) -> float:
95
+ if not required:
96
+ return 1.0
97
+ agent_text = " ".join(t["text"] for t in turns if t["speaker"] == "agent").lower()
98
+ hit = sum(1 for kw in required if str(kw).lower() in agent_text)
99
+ return hit / len(required)
100
+
101
+
102
+ def score_voice_episode(
103
+ dialogue: Sequence[Mapping[str, Any]],
104
+ *,
105
+ budgets: Mapping[str, Any] | None = None,
106
+ required_content: Sequence[str] | None = None,
107
+ ) -> dict[str, Any]:
108
+ """Score a voice episode transcript; return ``{"result", "raw"}`` (unified Result).
109
+
110
+ The scalar is the mean of the four sub-scores; the verdict (in pass_fail) is a
111
+ pass only if EVERY sub-score meets the floor — a single bad dimension (e.g. the
112
+ agent talks over the caller) fails the episode.
113
+ """
114
+
115
+ budgets = budgets or {}
116
+ required_content = required_content or []
117
+ if not dialogue:
118
+ return {
119
+ "result": {"scalar": 0.0, "components": {}, "pass_fail": {"voice": False},
120
+ "explanation": "empty transcript"},
121
+ "raw": {"modality": "voice"},
122
+ }
123
+ turns = _norm_turns(dialogue)
124
+ max_latency = int(budgets.get("max_latency_ms", _DEFAULT_MAX_LATENCY_MS))
125
+ latency = _latency_score(turns, max_latency)
126
+ turn_taking, bargein = _overlap_and_bargein(turns)
127
+ content = _content_score(turns, required_content)
128
+ sub = {
129
+ "latency": round(latency, 6),
130
+ "turn_taking": round(turn_taking, 6),
131
+ "barge_in": round(bargein, 6),
132
+ "content": round(content, 6),
133
+ }
134
+ scalar = round(sum(sub.values()) / len(sub), 6)
135
+ floors_met = all(v >= _PASS_FLOOR for v in sub.values())
136
+ return {
137
+ "result": {
138
+ "scalar": scalar,
139
+ "components": sub,
140
+ "pass_fail": {"voice": floors_met, **{f"{k}_floor": (v >= _PASS_FLOOR)
141
+ for k, v in sub.items()}},
142
+ "explanation": ("all voice dimensions met the floor" if floors_met
143
+ else "a voice dimension fell below the floor"),
144
+ },
145
+ "raw": {"modality": "voice", "turns": len(turns), "max_latency_ms": max_latency,
146
+ "floors_met": floors_met},
147
+ }