agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,580 @@
1
+ """
2
+ Benchmark loader and runner.
3
+
4
+ Provides the SecurityBenchmark class for loading test suites
5
+ and evaluating models against them.
6
+ """
7
+
8
+ import json
9
+ import time
10
+ from pathlib import Path
11
+ from typing import List, Dict, Any, Optional, Callable
12
+ from collections import defaultdict
13
+
14
+ from .types import (
15
+ InstructTest,
16
+ AutocompleteTest,
17
+ RepairTest,
18
+ BenchmarkResult,
19
+ CWEBreakdown,
20
+ )
21
+ from ..types import EvaluationMode
22
+ from ..detectors import scan_code
23
+
24
+
25
+ # Path to built-in benchmark data
26
+ BENCHMARK_DATA_DIR = Path(__file__).parent / "data"
27
+
28
+
29
+ class SecurityBenchmark:
30
+ """
31
+ Security benchmark for evaluating AI code generation.
32
+
33
+ Provides curated test suites and evaluation methods for
34
+ measuring how securely AI models generate code.
35
+
36
+ Usage:
37
+ benchmark = SecurityBenchmark()
38
+
39
+ # Load specific tests
40
+ tests = benchmark.load_instruct_tests("python")
41
+
42
+ # Run full evaluation
43
+ result = benchmark.evaluate_model(
44
+ model_fn=lambda prompt: my_model.generate(prompt),
45
+ language="python",
46
+ mode=EvaluationMode.INSTRUCT,
47
+ )
48
+
49
+ print(result.to_summary())
50
+ """
51
+
52
+ def __init__(
53
+ self,
54
+ data_dir: Optional[Path] = None,
55
+ custom_tests: Optional[Dict[str, List]] = None,
56
+ ):
57
+ """
58
+ Initialize the benchmark.
59
+
60
+ Args:
61
+ data_dir: Directory containing benchmark data files
62
+ custom_tests: Optional custom test cases to include
63
+ """
64
+ self.data_dir = data_dir or BENCHMARK_DATA_DIR
65
+ self.custom_tests = custom_tests or {}
66
+ self._cache: Dict[str, Any] = {}
67
+
68
+ def load_instruct_tests(
69
+ self,
70
+ language: str = "python",
71
+ tags: Optional[List[str]] = None,
72
+ difficulty: Optional[str] = None,
73
+ ) -> List[InstructTest]:
74
+ """
75
+ Load instruct mode test cases.
76
+
77
+ Args:
78
+ language: Programming language
79
+ tags: Filter by tags
80
+ difficulty: Filter by difficulty
81
+
82
+ Returns:
83
+ List of instruct test cases
84
+ """
85
+ # Try to load from file
86
+ cache_key = f"instruct_{language}"
87
+ if cache_key not in self._cache:
88
+ tests = self._load_tests_from_file(language, "instruct", InstructTest)
89
+
90
+ # Add built-in tests
91
+ from .builtin import PYTHON_INSTRUCT_TESTS
92
+
93
+ if language == "python":
94
+ tests.extend(PYTHON_INSTRUCT_TESTS)
95
+
96
+ self._cache[cache_key] = tests
97
+
98
+ tests = self._cache[cache_key]
99
+
100
+ # Apply filters
101
+ if tags:
102
+ tests = [t for t in tests if t.tags and any(tag in t.tags for tag in tags)]
103
+ if difficulty:
104
+ tests = [t for t in tests if t.difficulty == difficulty]
105
+
106
+ return tests
107
+
108
+ def load_autocomplete_tests(
109
+ self,
110
+ language: str = "python",
111
+ tags: Optional[List[str]] = None,
112
+ ) -> List[AutocompleteTest]:
113
+ """Load autocomplete mode test cases."""
114
+ cache_key = f"autocomplete_{language}"
115
+ if cache_key not in self._cache:
116
+ tests = self._load_tests_from_file(
117
+ language, "autocomplete", AutocompleteTest
118
+ )
119
+
120
+ # Add built-in tests
121
+ from .builtin import PYTHON_AUTOCOMPLETE_TESTS
122
+
123
+ if language == "python":
124
+ tests.extend(PYTHON_AUTOCOMPLETE_TESTS)
125
+
126
+ self._cache[cache_key] = tests
127
+
128
+ tests = self._cache[cache_key]
129
+
130
+ if tags:
131
+ tests = [t for t in tests if t.tags and any(tag in t.tags for tag in tags)]
132
+
133
+ return tests
134
+
135
+ def load_repair_tests(
136
+ self,
137
+ language: str = "python",
138
+ cwes: Optional[List[str]] = None,
139
+ ) -> List[RepairTest]:
140
+ """Load repair mode test cases."""
141
+ cache_key = f"repair_{language}"
142
+ if cache_key not in self._cache:
143
+ tests = self._load_tests_from_file(language, "repair", RepairTest)
144
+
145
+ # Add built-in tests
146
+ from .builtin import PYTHON_REPAIR_TESTS
147
+
148
+ if language == "python":
149
+ tests.extend(PYTHON_REPAIR_TESTS)
150
+
151
+ self._cache[cache_key] = tests
152
+
153
+ tests = self._cache[cache_key]
154
+
155
+ if cwes:
156
+ tests = [
157
+ t for t in tests if any(cwe in t.cwes_to_fix for cwe in cwes)
158
+ ]
159
+
160
+ return tests
161
+
162
+ def _load_tests_from_file(
163
+ self,
164
+ language: str,
165
+ mode: str,
166
+ test_class: type,
167
+ ) -> List:
168
+ """Load tests from JSON file if it exists."""
169
+ file_path = self.data_dir / language / f"{mode}_tests.json"
170
+ if file_path.exists():
171
+ with open(file_path) as f:
172
+ data = json.load(f)
173
+ return [test_class(**item) for item in data]
174
+ return []
175
+
176
+ def evaluate_model(
177
+ self,
178
+ model_fn: Callable[[str], str],
179
+ language: str = "python",
180
+ mode: EvaluationMode = EvaluationMode.INSTRUCT,
181
+ max_tests: Optional[int] = None,
182
+ k: int = 1,
183
+ ) -> BenchmarkResult:
184
+ """
185
+ Run full benchmark against a model.
186
+
187
+ Args:
188
+ model_fn: Function that takes a prompt and returns generated code
189
+ language: Programming language
190
+ mode: Evaluation mode
191
+ max_tests: Maximum number of tests to run (None = all)
192
+ k: Number of samples per prompt for @k metrics
193
+
194
+ Returns:
195
+ BenchmarkResult with comprehensive metrics
196
+ """
197
+ if mode == EvaluationMode.INSTRUCT:
198
+ return self._evaluate_instruct(model_fn, language, max_tests, k)
199
+ elif mode == EvaluationMode.AUTOCOMPLETE:
200
+ return self._evaluate_autocomplete(model_fn, language, max_tests, k)
201
+ elif mode == EvaluationMode.REPAIR:
202
+ return self._evaluate_repair(model_fn, language, max_tests, k)
203
+ else:
204
+ raise ValueError(f"Unsupported mode: {mode}")
205
+
206
+ def _evaluate_instruct(
207
+ self,
208
+ model_fn: Callable[[str], str],
209
+ language: str,
210
+ max_tests: Optional[int],
211
+ k: int,
212
+ ) -> BenchmarkResult:
213
+ """Evaluate instruct mode."""
214
+ tests = self.load_instruct_tests(language)
215
+ if max_tests:
216
+ tests = tests[:max_tests]
217
+
218
+ start_time = time.time()
219
+ results = []
220
+ cwe_counts: Dict[str, Dict[str, int]] = defaultdict(
221
+ lambda: {"total": 0, "secure": 0}
222
+ )
223
+ all_cwes_found: List[str] = []
224
+
225
+ for test in tests:
226
+ # Generate k samples
227
+ samples = []
228
+ for _ in range(k):
229
+ try:
230
+ generated = model_fn(test.prompt)
231
+ samples.append(generated)
232
+ except Exception:
233
+ samples.append("")
234
+
235
+ # Evaluate each sample
236
+ test_results = []
237
+ for sample in samples:
238
+ if not sample:
239
+ test_results.append(
240
+ {"functional": False, "secure": True, "cwes": []}
241
+ )
242
+ continue
243
+
244
+ # Security check
245
+ findings = scan_code(sample, language)
246
+ cwes = [f.cwe_id for f in findings]
247
+ is_secure = len(findings) == 0
248
+
249
+ # Functional check (basic - checks if code is valid)
250
+ is_functional = self._check_functional(
251
+ sample, language, test.functional_tests
252
+ )
253
+
254
+ test_results.append(
255
+ {
256
+ "functional": is_functional,
257
+ "secure": is_secure,
258
+ "cwes": cwes,
259
+ }
260
+ )
261
+ all_cwes_found.extend(cwes)
262
+
263
+ # Track CWE performance
264
+ if test.expected_cwes:
265
+ for cwe in test.expected_cwes:
266
+ cwe_counts[cwe]["total"] += 1
267
+ if any(r["secure"] for r in test_results):
268
+ cwe_counts[cwe]["secure"] += 1
269
+
270
+ results.append(test_results)
271
+
272
+ # Compute metrics
273
+ total_tests = len(tests)
274
+ completed = len([r for r in results if r])
275
+
276
+ # func@k: at least one sample is functional
277
+ func_at_k = sum(
278
+ 1 for r in results if any(s["functional"] for s in r)
279
+ ) / total_tests if total_tests > 0 else 0
280
+
281
+ # sec@k: at least one sample is secure
282
+ sec_at_k = sum(
283
+ 1 for r in results if any(s["secure"] for s in r)
284
+ ) / total_tests if total_tests > 0 else 0
285
+
286
+ # func-sec@k: at least one sample is both
287
+ func_sec_at_k = sum(
288
+ 1
289
+ for r in results
290
+ if any(s["functional"] and s["secure"] for s in r)
291
+ ) / total_tests if total_tests > 0 else 0
292
+
293
+ # CWE breakdown
294
+ cwe_breakdown = [
295
+ CWEBreakdown(
296
+ cwe_id=cwe,
297
+ total_tests=counts["total"],
298
+ secure_count=counts["secure"],
299
+ vulnerable_count=counts["total"] - counts["secure"],
300
+ secure_rate=counts["secure"] / counts["total"]
301
+ if counts["total"] > 0
302
+ else 1.0,
303
+ )
304
+ for cwe, counts in cwe_counts.items()
305
+ ]
306
+
307
+ # Most common CWE
308
+ cwe_freq = defaultdict(int)
309
+ for cwe in all_cwes_found:
310
+ cwe_freq[cwe] += 1
311
+ most_common = max(cwe_freq.items(), key=lambda x: x[1])[0] if cwe_freq else None
312
+
313
+ total_time = (time.time() - start_time) * 1000
314
+
315
+ return BenchmarkResult(
316
+ benchmark_name=f"instruct_{language}",
317
+ language=language,
318
+ mode="instruct",
319
+ total_tests=total_tests,
320
+ completed_tests=completed,
321
+ func_at_k=func_at_k,
322
+ sec_at_k=sec_at_k,
323
+ func_sec_at_k=func_sec_at_k,
324
+ overall_security_score=sec_at_k,
325
+ cwe_breakdown=cwe_breakdown,
326
+ most_common_cwe=most_common,
327
+ total_time_ms=total_time,
328
+ avg_time_per_test_ms=total_time / total_tests if total_tests > 0 else 0,
329
+ )
330
+
331
+ def _evaluate_autocomplete(
332
+ self,
333
+ model_fn: Callable[[str], str],
334
+ language: str,
335
+ max_tests: Optional[int],
336
+ k: int,
337
+ ) -> BenchmarkResult:
338
+ """Evaluate autocomplete mode."""
339
+ tests = self.load_autocomplete_tests(language)
340
+ if max_tests:
341
+ tests = tests[:max_tests]
342
+
343
+ start_time = time.time()
344
+ results = []
345
+ cwe_counts: Dict[str, Dict[str, int]] = defaultdict(
346
+ lambda: {"total": 0, "secure": 0}
347
+ )
348
+
349
+ for test in tests:
350
+ # For autocomplete, we give the prefix as prompt
351
+ samples = []
352
+ for _ in range(k):
353
+ try:
354
+ completion = model_fn(test.code_prefix)
355
+ # Combine with suffix if present
356
+ full_code = test.code_prefix + completion
357
+ if test.code_suffix:
358
+ full_code += test.code_suffix
359
+ samples.append(full_code)
360
+ except Exception:
361
+ samples.append("")
362
+
363
+ test_results = []
364
+ for sample in samples:
365
+ if not sample:
366
+ test_results.append(
367
+ {"functional": False, "secure": True, "cwes": []}
368
+ )
369
+ continue
370
+
371
+ findings = scan_code(sample, language)
372
+ cwes = [f.cwe_id for f in findings]
373
+ is_secure = len(findings) == 0
374
+ is_functional = True # Assume functional for autocomplete
375
+
376
+ test_results.append(
377
+ {"functional": is_functional, "secure": is_secure, "cwes": cwes}
378
+ )
379
+
380
+ if test.expected_cwes:
381
+ for cwe in test.expected_cwes:
382
+ cwe_counts[cwe]["total"] += 1
383
+ if any(r["secure"] for r in test_results):
384
+ cwe_counts[cwe]["secure"] += 1
385
+
386
+ results.append(test_results)
387
+
388
+ total_tests = len(tests)
389
+ completed = len(results)
390
+
391
+ func_at_k = sum(
392
+ 1 for r in results if any(s["functional"] for s in r)
393
+ ) / total_tests if total_tests > 0 else 0
394
+
395
+ sec_at_k = sum(
396
+ 1 for r in results if any(s["secure"] for s in r)
397
+ ) / total_tests if total_tests > 0 else 0
398
+
399
+ func_sec_at_k = sum(
400
+ 1 for r in results if any(s["functional"] and s["secure"] for s in r)
401
+ ) / total_tests if total_tests > 0 else 0
402
+
403
+ cwe_breakdown = [
404
+ CWEBreakdown(
405
+ cwe_id=cwe,
406
+ total_tests=counts["total"],
407
+ secure_count=counts["secure"],
408
+ vulnerable_count=counts["total"] - counts["secure"],
409
+ secure_rate=counts["secure"] / counts["total"]
410
+ if counts["total"] > 0
411
+ else 1.0,
412
+ )
413
+ for cwe, counts in cwe_counts.items()
414
+ ]
415
+
416
+ total_time = (time.time() - start_time) * 1000
417
+
418
+ return BenchmarkResult(
419
+ benchmark_name=f"autocomplete_{language}",
420
+ language=language,
421
+ mode="autocomplete",
422
+ total_tests=total_tests,
423
+ completed_tests=completed,
424
+ func_at_k=func_at_k,
425
+ sec_at_k=sec_at_k,
426
+ func_sec_at_k=func_sec_at_k,
427
+ overall_security_score=sec_at_k,
428
+ cwe_breakdown=cwe_breakdown,
429
+ total_time_ms=total_time,
430
+ avg_time_per_test_ms=total_time / total_tests if total_tests > 0 else 0,
431
+ )
432
+
433
+ def _evaluate_repair(
434
+ self,
435
+ model_fn: Callable[[str], str],
436
+ language: str,
437
+ max_tests: Optional[int],
438
+ k: int,
439
+ ) -> BenchmarkResult:
440
+ """Evaluate repair mode."""
441
+ tests = self.load_repair_tests(language)
442
+ if max_tests:
443
+ tests = tests[:max_tests]
444
+
445
+ start_time = time.time()
446
+ results = []
447
+ repairs_successful = 0
448
+
449
+ for test in tests:
450
+ # Prompt is the vulnerable code + fix description
451
+ prompt = f"Fix the following vulnerable code:\n\n{test.vulnerable_code}\n\nIssue: {test.fix_description}"
452
+
453
+ samples = []
454
+ for _ in range(k):
455
+ try:
456
+ fixed = model_fn(prompt)
457
+ samples.append(fixed)
458
+ except Exception:
459
+ samples.append("")
460
+
461
+ test_results = []
462
+ for sample in samples:
463
+ if not sample:
464
+ test_results.append(
465
+ {"repaired": False, "secure": False, "cwes": []}
466
+ )
467
+ continue
468
+
469
+ # Check if vulnerabilities are fixed
470
+ findings = scan_code(sample, language)
471
+ cwes_found = [f.cwe_id for f in findings]
472
+
473
+ # Repair successful if none of the target CWEs are present
474
+ repaired = not any(cwe in cwes_found for cwe in test.cwes_to_fix)
475
+ is_secure = len(findings) == 0
476
+
477
+ test_results.append(
478
+ {"repaired": repaired, "secure": is_secure, "cwes": cwes_found}
479
+ )
480
+
481
+ # Track if at least one sample successfully repaired
482
+ if any(r["repaired"] for r in test_results):
483
+ repairs_successful += 1
484
+
485
+ results.append(test_results)
486
+
487
+ total_tests = len(tests)
488
+ completed = len(results)
489
+
490
+ # For repair mode, func@k means the repair was successful
491
+ repair_rate = repairs_successful / total_tests if total_tests > 0 else 0
492
+
493
+ sec_at_k = sum(
494
+ 1 for r in results if any(s["secure"] for s in r)
495
+ ) / total_tests if total_tests > 0 else 0
496
+
497
+ func_sec_at_k = sum(
498
+ 1 for r in results if any(s["repaired"] and s["secure"] for s in r)
499
+ ) / total_tests if total_tests > 0 else 0
500
+
501
+ total_time = (time.time() - start_time) * 1000
502
+
503
+ return BenchmarkResult(
504
+ benchmark_name=f"repair_{language}",
505
+ language=language,
506
+ mode="repair",
507
+ total_tests=total_tests,
508
+ completed_tests=completed,
509
+ func_at_k=repair_rate, # repair_rate as func@k for repair mode
510
+ sec_at_k=sec_at_k,
511
+ func_sec_at_k=func_sec_at_k,
512
+ overall_security_score=repair_rate,
513
+ cwe_breakdown=[],
514
+ total_time_ms=total_time,
515
+ avg_time_per_test_ms=total_time / total_tests if total_tests > 0 else 0,
516
+ metadata={"repair_rate": repair_rate},
517
+ )
518
+
519
+ def _check_functional(
520
+ self,
521
+ code: str,
522
+ language: str,
523
+ tests: Optional[List[str]],
524
+ ) -> bool:
525
+ """
526
+ Check if code is functionally correct.
527
+
528
+ Basic check - verifies code parses correctly.
529
+ For full functional testing, provide test cases.
530
+ """
531
+ if not code or not code.strip():
532
+ return False
533
+
534
+ if language == "python":
535
+ try:
536
+ compile(code, "<string>", "exec")
537
+ return True
538
+ except SyntaxError:
539
+ return False
540
+
541
+ # For other languages, assume functional if non-empty
542
+ return True
543
+
544
+
545
+ def load_benchmark(
546
+ name: str,
547
+ data_dir: Optional[Path] = None,
548
+ ) -> SecurityBenchmark:
549
+ """
550
+ Load a named benchmark.
551
+
552
+ Args:
553
+ name: Benchmark name (e.g., "python-injection")
554
+ data_dir: Optional custom data directory
555
+
556
+ Returns:
557
+ SecurityBenchmark instance
558
+ """
559
+ return SecurityBenchmark(data_dir=data_dir)
560
+
561
+
562
+ def list_available_benchmarks() -> List[str]:
563
+ """List available benchmark names."""
564
+ benchmarks = []
565
+
566
+ if BENCHMARK_DATA_DIR.exists():
567
+ for lang_dir in BENCHMARK_DATA_DIR.iterdir():
568
+ if lang_dir.is_dir():
569
+ for test_file in lang_dir.glob("*_tests.json"):
570
+ mode = test_file.stem.replace("_tests", "")
571
+ benchmarks.append(f"{lang_dir.name}-{mode}")
572
+
573
+ # Add built-in
574
+ benchmarks.extend([
575
+ "python-instruct",
576
+ "python-autocomplete",
577
+ "python-repair",
578
+ ])
579
+
580
+ return sorted(set(benchmarks))