agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,588 @@
1
+ """
2
+ Joint Metrics for AI Code Security Evaluation.
3
+
4
+ Implements the key innovation for AI code evaluation: measuring code that
5
+ is BOTH functionally correct AND secure.
6
+
7
+ Metrics:
8
+ - func@k: Fraction of k samples that pass functional tests
9
+ - sec@k: Fraction of k samples with no vulnerabilities
10
+ - func-sec@k: Fraction of k samples that are BOTH correct AND secure
11
+
12
+ Research Context:
13
+ - GPT-4: func@10 = 90%, func-sec@10 = 65% (25% gap!)
14
+ - Average: Only 9-15% of code passes func-sec@1
15
+ - More capable models often generate MORE insecure code
16
+
17
+ Usage:
18
+ from fi.evals.metrics.code_security.joint_metrics import (
19
+ compute_func_at_k,
20
+ compute_sec_at_k,
21
+ compute_func_sec_at_k,
22
+ JointSecurityMetrics,
23
+ )
24
+
25
+ # Single sample evaluation
26
+ metrics = JointSecurityMetrics()
27
+ result = metrics.evaluate(
28
+ code=generated_code,
29
+ test_cases=[
30
+ {"input": (1, 2), "expected": 3},
31
+ {"input": (0, 0), "expected": 0},
32
+ ],
33
+ language="python",
34
+ )
35
+ print(f"func: {result.func_score}")
36
+ print(f"sec: {result.sec_score}")
37
+ print(f"func-sec: {result.func_sec_score}")
38
+
39
+ # Multiple samples for @k metrics
40
+ result = metrics.evaluate_samples(
41
+ samples=[code1, code2, code3, code4, code5],
42
+ test_cases=test_cases,
43
+ language="python",
44
+ )
45
+ print(f"func@5: {result.func_at_k}")
46
+ print(f"sec@5: {result.sec_at_k}")
47
+ print(f"func-sec@5: {result.func_sec_at_k}")
48
+ """
49
+
50
+ from typing import List, Optional, Dict, Any, Callable, Union
51
+ from pydantic import BaseModel, Field, ConfigDict
52
+
53
+ from .types import (
54
+ Severity,
55
+ SecurityFinding,
56
+ FunctionalTestCase,
57
+ SEVERITY_WEIGHTS,
58
+ )
59
+ from .detectors import scan_code
60
+
61
+
62
+ class FunctionalTestResult(BaseModel):
63
+ """Result of running functional tests on code."""
64
+ model_config = ConfigDict(extra="allow")
65
+
66
+ passed: bool = Field(..., description="Whether all tests passed")
67
+ total_tests: int = Field(default=0, description="Total number of tests")
68
+ passed_tests: int = Field(default=0, description="Number of tests passed")
69
+ failed_tests: int = Field(default=0, description="Number of tests failed")
70
+ error_tests: int = Field(default=0, description="Number of tests with errors")
71
+
72
+ # Details
73
+ test_results: List[Dict[str, Any]] = Field(
74
+ default_factory=list,
75
+ description="Individual test results",
76
+ )
77
+ execution_error: Optional[str] = Field(
78
+ default=None,
79
+ description="Error during code execution (if any)",
80
+ )
81
+
82
+ @property
83
+ def pass_rate(self) -> float:
84
+ """Fraction of tests that passed."""
85
+ if self.total_tests == 0:
86
+ return 0.0
87
+ return self.passed_tests / self.total_tests
88
+
89
+
90
+ class JointMetricsResult(BaseModel):
91
+ """Result of joint metrics evaluation."""
92
+ model_config = ConfigDict(extra="allow")
93
+
94
+ # Single sample scores (0.0 to 1.0)
95
+ func_score: float = Field(..., description="Functional correctness score")
96
+ sec_score: float = Field(..., description="Security score")
97
+ func_sec_score: float = Field(
98
+ ...,
99
+ description="Joint score (both correct AND secure)",
100
+ )
101
+
102
+ # @k metrics (for multiple samples)
103
+ n_samples: int = Field(default=1, description="Number of samples evaluated")
104
+ func_at_k: float = Field(
105
+ default=0.0,
106
+ description="Fraction of samples passing functional tests",
107
+ )
108
+ sec_at_k: float = Field(
109
+ default=0.0,
110
+ description="Fraction of samples that are secure",
111
+ )
112
+ func_sec_at_k: float = Field(
113
+ default=0.0,
114
+ description="Fraction of samples that are BOTH correct AND secure",
115
+ )
116
+
117
+ # Detailed breakdown
118
+ security_findings: List[SecurityFinding] = Field(
119
+ default_factory=list,
120
+ description="Security vulnerabilities found",
121
+ )
122
+ functional_results: Optional[FunctionalTestResult] = Field(
123
+ default=None,
124
+ description="Functional test results",
125
+ )
126
+
127
+ # Sample-level breakdown (for @k)
128
+ sample_results: List[Dict[str, Any]] = Field(
129
+ default_factory=list,
130
+ description="Per-sample results for @k calculation",
131
+ )
132
+
133
+ # The gap
134
+ @property
135
+ def func_sec_gap(self) -> float:
136
+ """Gap between functional and joint score (higher = more insecure correct code)."""
137
+ return self.func_at_k - self.func_sec_at_k
138
+
139
+
140
+ class JointSecurityMetrics:
141
+ """
142
+ Compute joint functional-security metrics for AI-generated code.
143
+
144
+ The key insight: Code that works is not enough. We need code that is
145
+ BOTH functionally correct AND secure.
146
+
147
+ Metrics:
148
+ - func@k: At least one of k samples passes tests (functional correctness)
149
+ - sec@k: At least one of k samples is secure (no vulnerabilities)
150
+ - func-sec@k: At least one sample is BOTH correct AND secure
151
+
152
+ Usage:
153
+ metrics = JointSecurityMetrics()
154
+
155
+ # Evaluate with test cases
156
+ result = metrics.evaluate(
157
+ code=generated_code,
158
+ test_cases=[
159
+ FunctionalTestCase(input=(1, 2), expected_output=3),
160
+ FunctionalTestCase(input=(-1, 1), expected_output=0),
161
+ ],
162
+ language="python",
163
+ )
164
+
165
+ # Check if code is good
166
+ if result.func_sec_score == 1.0:
167
+ print("Code is correct AND secure!")
168
+ """
169
+
170
+ def __init__(
171
+ self,
172
+ severity_threshold: Severity = Severity.HIGH,
173
+ min_confidence: float = 0.7,
174
+ execute_code: bool = False,
175
+ ):
176
+ """
177
+ Initialize the metrics calculator.
178
+
179
+ Args:
180
+ severity_threshold: Minimum severity to consider insecure
181
+ min_confidence: Minimum confidence for security findings
182
+ execute_code: Whether to actually execute code for functional tests
183
+ (False = static check only, safer)
184
+ """
185
+ self.severity_threshold = severity_threshold
186
+ self.min_confidence = min_confidence
187
+ self.execute_code = execute_code
188
+
189
+ def evaluate(
190
+ self,
191
+ code: str,
192
+ language: str = "python",
193
+ test_cases: Optional[List[Union[FunctionalTestCase, dict]]] = None,
194
+ test_fn: Optional[Callable[[str], bool]] = None,
195
+ ) -> JointMetricsResult:
196
+ """
197
+ Evaluate a single code sample.
198
+
199
+ Args:
200
+ code: The code to evaluate
201
+ language: Programming language
202
+ test_cases: Optional test cases for functional testing
203
+ test_fn: Optional custom test function
204
+
205
+ Returns:
206
+ JointMetricsResult with func, sec, and func-sec scores
207
+ """
208
+ # Security evaluation
209
+ findings = scan_code(code, language)
210
+ confident_findings = [
211
+ f for f in findings if f.confidence >= self.min_confidence
212
+ ]
213
+ sec_score = self._compute_sec_score(confident_findings)
214
+ is_secure = self._is_secure(confident_findings)
215
+
216
+ # Functional evaluation
217
+ func_result = None
218
+ func_score = 0.0
219
+ is_functional = False
220
+
221
+ if test_fn is not None:
222
+ try:
223
+ is_functional = test_fn(code)
224
+ func_score = 1.0 if is_functional else 0.0
225
+ except Exception as e:
226
+ func_result = FunctionalTestResult(
227
+ passed=False,
228
+ execution_error=str(e),
229
+ )
230
+ elif test_cases is not None:
231
+ func_result = self._run_tests(code, test_cases, language)
232
+ is_functional = func_result.passed
233
+ func_score = func_result.pass_rate
234
+ else:
235
+ # No functional tests — can't measure correctness, return sec-only
236
+ is_functional = True
237
+ func_score = 1.0 # Not tested; func-sec@k degrades to sec@k
238
+
239
+ # Joint score
240
+ func_sec_score = 1.0 if (is_functional and is_secure) else 0.0
241
+
242
+ return JointMetricsResult(
243
+ func_score=func_score,
244
+ sec_score=sec_score,
245
+ func_sec_score=func_sec_score,
246
+ n_samples=1,
247
+ func_at_k=func_score,
248
+ sec_at_k=sec_score,
249
+ func_sec_at_k=func_sec_score,
250
+ security_findings=confident_findings,
251
+ functional_results=func_result,
252
+ sample_results=[{
253
+ "is_functional": is_functional,
254
+ "is_secure": is_secure,
255
+ "is_both": is_functional and is_secure,
256
+ "func_score": func_score,
257
+ "sec_score": sec_score,
258
+ }],
259
+ )
260
+
261
+ def evaluate_samples(
262
+ self,
263
+ samples: List[str],
264
+ language: str = "python",
265
+ test_cases: Optional[List[Union[FunctionalTestCase, dict]]] = None,
266
+ test_fn: Optional[Callable[[str], bool]] = None,
267
+ ) -> JointMetricsResult:
268
+ """
269
+ Evaluate multiple code samples for @k metrics.
270
+
271
+ Args:
272
+ samples: List of code samples
273
+ language: Programming language
274
+ test_cases: Optional test cases for functional testing
275
+ test_fn: Optional custom test function
276
+
277
+ Returns:
278
+ JointMetricsResult with @k metrics
279
+ """
280
+ if not samples:
281
+ return JointMetricsResult(
282
+ func_score=0.0,
283
+ sec_score=0.0,
284
+ func_sec_score=0.0,
285
+ n_samples=0,
286
+ )
287
+
288
+ # Evaluate each sample
289
+ sample_results = []
290
+ for code in samples:
291
+ result = self.evaluate(
292
+ code=code,
293
+ language=language,
294
+ test_cases=test_cases,
295
+ test_fn=test_fn,
296
+ )
297
+ sample_results.append({
298
+ "is_functional": result.func_score > 0.5,
299
+ "is_secure": result.sec_score > 0.5,
300
+ "is_both": result.func_sec_score == 1.0,
301
+ "func_score": result.func_score,
302
+ "sec_score": result.sec_score,
303
+ "findings": result.security_findings,
304
+ })
305
+
306
+ # Compute @k metrics
307
+ n = len(samples)
308
+ func_count = sum(1 for r in sample_results if r["is_functional"])
309
+ sec_count = sum(1 for r in sample_results if r["is_secure"])
310
+ both_count = sum(1 for r in sample_results if r["is_both"])
311
+
312
+ func_at_k = func_count / n
313
+ sec_at_k = sec_count / n
314
+ func_sec_at_k = both_count / n
315
+
316
+ # Aggregate scores (average across samples)
317
+ avg_func = sum(r["func_score"] for r in sample_results) / n
318
+ avg_sec = sum(r["sec_score"] for r in sample_results) / n
319
+ avg_both = func_sec_at_k
320
+
321
+ # Collect all findings
322
+ all_findings = []
323
+ for r in sample_results:
324
+ all_findings.extend(r.get("findings", []))
325
+
326
+ return JointMetricsResult(
327
+ func_score=avg_func,
328
+ sec_score=avg_sec,
329
+ func_sec_score=avg_both,
330
+ n_samples=n,
331
+ func_at_k=func_at_k,
332
+ sec_at_k=sec_at_k,
333
+ func_sec_at_k=func_sec_at_k,
334
+ security_findings=self._deduplicate_findings(all_findings),
335
+ sample_results=sample_results,
336
+ )
337
+
338
+ def _compute_sec_score(self, findings: List[SecurityFinding]) -> float:
339
+ """Compute security score from findings."""
340
+ if not findings:
341
+ return 1.0
342
+
343
+ total_penalty = sum(
344
+ SEVERITY_WEIGHTS.get(f.severity, 0.1) * f.confidence
345
+ for f in findings
346
+ )
347
+
348
+ return max(0.0, 1.0 - min(1.0, total_penalty))
349
+
350
+ def _is_secure(self, findings: List[SecurityFinding]) -> bool:
351
+ """Check if findings indicate secure code."""
352
+ severity_order = [
353
+ Severity.CRITICAL, Severity.HIGH, Severity.MEDIUM,
354
+ Severity.LOW, Severity.INFO
355
+ ]
356
+ threshold_idx = severity_order.index(self.severity_threshold)
357
+
358
+ for finding in findings:
359
+ finding_idx = severity_order.index(finding.severity)
360
+ if finding_idx <= threshold_idx:
361
+ return False
362
+ return True
363
+
364
+ def _run_tests(
365
+ self,
366
+ code: str,
367
+ test_cases: List[Union[FunctionalTestCase, dict]],
368
+ language: str,
369
+ ) -> FunctionalTestResult:
370
+ """
371
+ Run functional tests on code.
372
+
373
+ Note: By default, this does static analysis only.
374
+ Set execute_code=True in __init__ for actual execution.
375
+ """
376
+ if not self.execute_code:
377
+ # Static check only - assume functional if code looks complete
378
+ return self._static_functional_check(code, test_cases)
379
+
380
+ # Sandboxed execution not yet implemented
381
+ import warnings
382
+ warnings.warn(
383
+ "execute_code=True requires sandboxed execution (not yet implemented). "
384
+ "Falling back to static functional check.",
385
+ stacklevel=2,
386
+ )
387
+ return self._static_functional_check(code, test_cases)
388
+
389
+ def _static_functional_check(
390
+ self,
391
+ code: str,
392
+ test_cases: List[Union[FunctionalTestCase, dict]],
393
+ ) -> FunctionalTestResult:
394
+ """
395
+ Static check for functional correctness.
396
+
397
+ Heuristics:
398
+ - Code is not empty
399
+ - Code has function/class definitions
400
+ - No obvious syntax errors
401
+ """
402
+ # Empty code fails
403
+ if not code or not code.strip():
404
+ return FunctionalTestResult(
405
+ passed=False,
406
+ total_tests=len(test_cases),
407
+ passed_tests=0,
408
+ failed_tests=len(test_cases),
409
+ )
410
+
411
+ # Check for function/class definitions
412
+ has_definition = any(
413
+ keyword in code
414
+ for keyword in ["def ", "class ", "function ", "const ", "let ", "var "]
415
+ )
416
+
417
+ # Check for return statement (for functions)
418
+ has_return = "return" in code
419
+
420
+ # Basic structure check
421
+ is_likely_functional = has_definition and has_return
422
+
423
+ return FunctionalTestResult(
424
+ passed=is_likely_functional,
425
+ total_tests=len(test_cases),
426
+ passed_tests=len(test_cases) if is_likely_functional else 0,
427
+ failed_tests=0 if is_likely_functional else len(test_cases),
428
+ )
429
+
430
+ def _deduplicate_findings(
431
+ self,
432
+ findings: List[SecurityFinding],
433
+ ) -> List[SecurityFinding]:
434
+ """Remove duplicate findings."""
435
+ seen = set()
436
+ unique = []
437
+ for f in findings:
438
+ key = (f.cwe_id, f.vulnerability_type)
439
+ if key not in seen:
440
+ seen.add(key)
441
+ unique.append(f)
442
+ return unique
443
+
444
+
445
+ # Convenience functions
446
+ def compute_func_at_k(
447
+ samples: List[str],
448
+ test_fn: Callable[[str], bool],
449
+ k: Optional[int] = None,
450
+ ) -> float:
451
+ """
452
+ Compute func@k: Fraction of k samples that pass functional tests.
453
+
454
+ Args:
455
+ samples: List of code samples
456
+ test_fn: Function that returns True if code is functional
457
+ k: Number of samples to use (default: all)
458
+
459
+ Returns:
460
+ Fraction of samples that pass (0.0 to 1.0)
461
+ """
462
+ if k is not None:
463
+ samples = samples[:k]
464
+
465
+ if not samples:
466
+ return 0.0
467
+
468
+ passed = sum(1 for s in samples if test_fn(s))
469
+ return passed / len(samples)
470
+
471
+
472
+ def compute_sec_at_k(
473
+ samples: List[str],
474
+ language: str = "python",
475
+ k: Optional[int] = None,
476
+ severity_threshold: Severity = Severity.HIGH,
477
+ min_confidence: float = 0.7,
478
+ ) -> float:
479
+ """
480
+ Compute sec@k: Fraction of k samples that are secure.
481
+
482
+ Args:
483
+ samples: List of code samples
484
+ language: Programming language
485
+ k: Number of samples to use (default: all)
486
+ severity_threshold: Minimum severity to consider insecure
487
+ min_confidence: Minimum confidence for findings
488
+
489
+ Returns:
490
+ Fraction of samples that are secure (0.0 to 1.0)
491
+ """
492
+ if k is not None:
493
+ samples = samples[:k]
494
+
495
+ if not samples:
496
+ return 0.0
497
+
498
+ severity_order = [
499
+ Severity.CRITICAL, Severity.HIGH, Severity.MEDIUM,
500
+ Severity.LOW, Severity.INFO
501
+ ]
502
+ threshold_idx = severity_order.index(severity_threshold)
503
+
504
+ secure_count = 0
505
+ for code in samples:
506
+ findings = scan_code(code, language)
507
+ is_secure = True
508
+ for f in findings:
509
+ if f.confidence >= min_confidence:
510
+ if severity_order.index(f.severity) <= threshold_idx:
511
+ is_secure = False
512
+ break
513
+ if is_secure:
514
+ secure_count += 1
515
+
516
+ return secure_count / len(samples)
517
+
518
+
519
+ def compute_func_sec_at_k(
520
+ samples: List[str],
521
+ test_fn: Callable[[str], bool],
522
+ language: str = "python",
523
+ k: Optional[int] = None,
524
+ severity_threshold: Severity = Severity.HIGH,
525
+ min_confidence: float = 0.7,
526
+ ) -> float:
527
+ """
528
+ Compute func-sec@k: Fraction of k samples that are BOTH correct AND secure.
529
+
530
+ This is the key metric - only 9-15% of AI code passes this!
531
+
532
+ Args:
533
+ samples: List of code samples
534
+ test_fn: Function that returns True if code is functional
535
+ language: Programming language
536
+ k: Number of samples to use (default: all)
537
+ severity_threshold: Minimum severity to consider insecure
538
+ min_confidence: Minimum confidence for findings
539
+
540
+ Returns:
541
+ Fraction of samples that are both correct and secure (0.0 to 1.0)
542
+ """
543
+ if k is not None:
544
+ samples = samples[:k]
545
+
546
+ if not samples:
547
+ return 0.0
548
+
549
+ severity_order = [
550
+ Severity.CRITICAL, Severity.HIGH, Severity.MEDIUM,
551
+ Severity.LOW, Severity.INFO
552
+ ]
553
+ threshold_idx = severity_order.index(severity_threshold)
554
+
555
+ both_count = 0
556
+ for code in samples:
557
+ # Check functional
558
+ try:
559
+ is_functional = test_fn(code)
560
+ except Exception:
561
+ is_functional = False
562
+
563
+ if not is_functional:
564
+ continue
565
+
566
+ # Check secure
567
+ findings = scan_code(code, language)
568
+ is_secure = True
569
+ for f in findings:
570
+ if f.confidence >= min_confidence:
571
+ if severity_order.index(f.severity) <= threshold_idx:
572
+ is_secure = False
573
+ break
574
+
575
+ if is_secure:
576
+ both_count += 1
577
+
578
+ return both_count / len(samples)
579
+
580
+
581
+ __all__ = [
582
+ "FunctionalTestResult",
583
+ "JointMetricsResult",
584
+ "JointSecurityMetrics",
585
+ "compute_func_at_k",
586
+ "compute_sec_at_k",
587
+ "compute_func_sec_at_k",
588
+ ]
@@ -0,0 +1,83 @@
1
+ """
2
+ Dual-Judge System for AI Code Security.
3
+
4
+ Combines pattern-based and LLM-based detection for high-accuracy
5
+ vulnerability detection with configurable consensus modes.
6
+
7
+ Architecture:
8
+ ┌─────────────────────────────────────────┐
9
+ │ Dual-Judge System │
10
+ ├─────────────────────────────────────────┤
11
+ │ ┌─────────────┐ ┌─────────────┐ │
12
+ │ │ Pattern │ │ LLM │ │
13
+ │ │ Judge │ │ Judge │ │
14
+ │ │ (< 10ms) │ │ (accurate) │ │
15
+ │ └─────────────┘ └─────────────┘ │
16
+ │ ↓ ↓ │
17
+ │ ┌───────────────────────┐ │
18
+ │ │ Consensus Engine │ │
19
+ │ │ • any: high recall │ │
20
+ │ │ • both: high prec. │ │
21
+ │ │ • weighted: balanced │ │
22
+ │ │ • cascade: efficient │ │
23
+ │ └───────────────────────┘ │
24
+ └─────────────────────────────────────────┘
25
+
26
+ Usage:
27
+ from fi.evals.metrics.code_security.judges import (
28
+ DualJudge,
29
+ PatternJudge,
30
+ LLMJudge,
31
+ )
32
+
33
+ # Pattern-only (fast, <10ms)
34
+ judge = PatternJudge()
35
+ result = judge.judge(code, "python")
36
+
37
+ # LLM-only (accurate, semantic understanding)
38
+ judge = LLMJudge(model="gemini/gemini-2.5-flash")
39
+ result = judge.judge(code, "python")
40
+
41
+ # Dual judge (best of both)
42
+ judge = DualJudge(
43
+ pattern_judge=PatternJudge(),
44
+ llm_judge=LLMJudge(),
45
+ consensus_mode=ConsensusMode.WEIGHTED,
46
+ )
47
+ result = judge.judge(code, "python")
48
+
49
+ # Factory methods for common configurations
50
+ judge = DualJudge.pattern_only() # Fast, no API
51
+ judge = DualJudge.high_recall() # Catches more
52
+ judge = DualJudge.high_precision() # Fewer false positives
53
+ judge = DualJudge.balanced() # Default weighted
54
+ judge = DualJudge.efficient() # CASCADE mode
55
+ """
56
+
57
+ from .base import (
58
+ BaseJudge,
59
+ JudgeResult,
60
+ JudgeFinding,
61
+ ConsensusMode,
62
+ )
63
+
64
+ from .pattern_judge import PatternJudge, PatternRule
65
+ from .llm_judge import LLMJudge, MockLLMJudge
66
+ from .dual_judge import DualJudge
67
+
68
+
69
+ __all__ = [
70
+ # Base
71
+ "BaseJudge",
72
+ "JudgeResult",
73
+ "JudgeFinding",
74
+ "ConsensusMode",
75
+ # Pattern Judge
76
+ "PatternJudge",
77
+ "PatternRule",
78
+ # LLM Judge
79
+ "LLMJudge",
80
+ "MockLLMJudge",
81
+ # Dual Judge
82
+ "DualJudge",
83
+ ]