agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,534 @@
1
+ """
2
+ Dual-Judge System for AI Code Security.
3
+
4
+ Combines pattern-based and LLM-based detection with configurable
5
+ consensus modes for optimal precision/recall tradeoffs.
6
+ """
7
+
8
+ import time
9
+ from typing import List, Dict, Any, Optional, Tuple
10
+ from concurrent.futures import ThreadPoolExecutor, TimeoutError
11
+
12
+ from .base import BaseJudge, JudgeResult, JudgeFinding, ConsensusMode
13
+ from .pattern_judge import PatternJudge
14
+ from .llm_judge import LLMJudge
15
+ from ..types import Severity
16
+
17
+
18
+ class DualJudge(BaseJudge):
19
+ """
20
+ Dual-judge system combining pattern and LLM analysis.
21
+
22
+ Provides multiple consensus modes to balance precision and recall:
23
+
24
+ - ANY: Flag if either judge flags (high recall, may have false positives)
25
+ - BOTH: Flag only if both agree (high precision, may miss some)
26
+ - WEIGHTED: Weighted combination of confidences (balanced)
27
+ - CASCADE: Pattern first, LLM only for uncertain cases (efficient)
28
+
29
+ Architecture:
30
+ ┌─────────────────────────────────────────┐
31
+ │ Dual-Judge System │
32
+ ├─────────────────────────────────────────┤
33
+ │ ┌─────────────┐ ┌─────────────┐ │
34
+ │ │ Pattern │ │ LLM │ │
35
+ │ │ Judge │ │ Judge │ │
36
+ │ │ (< 10ms) │ │ (accurate) │ │
37
+ │ └─────────────┘ └─────────────┘ │
38
+ │ ↓ ↓ │
39
+ │ ┌───────────────────────┐ │
40
+ │ │ Consensus Engine │ │
41
+ │ │ • any: high recall │ │
42
+ │ │ • both: high prec. │ │
43
+ │ │ • weighted: balanced │ │
44
+ │ │ • cascade: efficient │ │
45
+ │ └───────────────────────┘ │
46
+ └─────────────────────────────────────────┘
47
+
48
+ Usage:
49
+ # Default dual judge
50
+ judge = DualJudge()
51
+ result = judge.judge(code, "python")
52
+
53
+ # High precision mode
54
+ judge = DualJudge(consensus_mode=ConsensusMode.BOTH)
55
+
56
+ # High recall mode
57
+ judge = DualJudge(consensus_mode=ConsensusMode.ANY)
58
+
59
+ # Efficient cascade mode
60
+ judge = DualJudge(consensus_mode=ConsensusMode.CASCADE)
61
+
62
+ # Custom judges
63
+ judge = DualJudge(
64
+ pattern_judge=PatternJudge.with_strict_rules(),
65
+ llm_judge=LLMJudge.with_gpt4(),
66
+ consensus_mode=ConsensusMode.WEIGHTED,
67
+ )
68
+ """
69
+
70
+ judge_type = "dual"
71
+
72
+ def __init__(
73
+ self,
74
+ pattern_judge: Optional[PatternJudge] = None,
75
+ llm_judge: Optional[LLMJudge] = None,
76
+ consensus_mode: ConsensusMode = ConsensusMode.WEIGHTED,
77
+ severity_threshold: Severity = Severity.HIGH,
78
+ min_confidence: float = 0.7,
79
+ pattern_weight: float = 0.4,
80
+ llm_weight: float = 0.6,
81
+ cascade_threshold: float = 0.6,
82
+ parallel: bool = True,
83
+ llm_timeout: float = 30.0,
84
+ ):
85
+ """
86
+ Initialize the dual judge.
87
+
88
+ Args:
89
+ pattern_judge: Pattern-based judge (default: PatternJudge())
90
+ llm_judge: LLM-based judge (default: None, pattern-only)
91
+ consensus_mode: How to combine judge results
92
+ severity_threshold: Minimum severity to flag as insecure
93
+ min_confidence: Minimum confidence to include findings
94
+ pattern_weight: Weight for pattern judge in WEIGHTED mode
95
+ llm_weight: Weight for LLM judge in WEIGHTED mode
96
+ cascade_threshold: Confidence threshold for CASCADE mode
97
+ parallel: Run judges in parallel
98
+ llm_timeout: Timeout for LLM judge
99
+ """
100
+ super().__init__(severity_threshold, min_confidence)
101
+
102
+ self.pattern_judge = pattern_judge or PatternJudge(
103
+ severity_threshold=severity_threshold,
104
+ min_confidence=min_confidence,
105
+ )
106
+ self.llm_judge = llm_judge
107
+ self.consensus_mode = consensus_mode
108
+ self.pattern_weight = pattern_weight
109
+ self.llm_weight = llm_weight
110
+ self.cascade_threshold = cascade_threshold
111
+ self.parallel = parallel
112
+ self.llm_timeout = llm_timeout
113
+
114
+ def judge(
115
+ self,
116
+ code: str,
117
+ language: str = "python",
118
+ context: Optional[Dict[str, Any]] = None,
119
+ ) -> JudgeResult:
120
+ """
121
+ Judge code using dual-judge system.
122
+
123
+ Args:
124
+ code: Source code to analyze
125
+ language: Programming language
126
+ context: Optional context
127
+
128
+ Returns:
129
+ JudgeResult with combined findings
130
+ """
131
+ start_time = time.time()
132
+
133
+ # Get pattern result (always run)
134
+ pattern_result = self.pattern_judge.judge(code, language, context)
135
+
136
+ # Get LLM result if configured
137
+ llm_result = None
138
+ if self.llm_judge is not None:
139
+ if self.consensus_mode == ConsensusMode.CASCADE:
140
+ # Only run LLM if pattern result is uncertain
141
+ if self._should_cascade(pattern_result):
142
+ llm_result = self._run_llm_judge(code, language, context)
143
+ elif self.parallel:
144
+ # Run in parallel (but we already have pattern result)
145
+ llm_result = self._run_llm_judge(code, language, context)
146
+ else:
147
+ llm_result = self._run_llm_judge(code, language, context)
148
+
149
+ # Combine results based on consensus mode
150
+ combined_result = self._combine_results(
151
+ pattern_result, llm_result, language
152
+ )
153
+
154
+ execution_time = (time.time() - start_time) * 1000
155
+ combined_result.execution_time_ms = execution_time
156
+
157
+ return combined_result
158
+
159
+ def _run_llm_judge(
160
+ self,
161
+ code: str,
162
+ language: str,
163
+ context: Optional[Dict[str, Any]],
164
+ ) -> Optional[JudgeResult]:
165
+ """Run LLM judge with timeout handling."""
166
+ if self.llm_judge is None:
167
+ return None
168
+
169
+ try:
170
+ with ThreadPoolExecutor(max_workers=1) as executor:
171
+ future = executor.submit(
172
+ self.llm_judge.judge, code, language, context
173
+ )
174
+ return future.result(timeout=self.llm_timeout)
175
+ except TimeoutError:
176
+ return None
177
+ except Exception:
178
+ return None
179
+
180
+ def _should_cascade(self, pattern_result: JudgeResult) -> bool:
181
+ """Determine if LLM should be invoked in CASCADE mode."""
182
+ # Invoke LLM if:
183
+ # 1. Pattern found findings but confidence is moderate
184
+ # 2. Pattern found no findings (might be false negative)
185
+
186
+ if not pattern_result.findings:
187
+ # No findings - let LLM check for false negatives
188
+ return True
189
+
190
+ # Check if any findings have moderate confidence
191
+ avg_confidence = sum(f.confidence for f in pattern_result.findings) / len(
192
+ pattern_result.findings
193
+ )
194
+ return avg_confidence < self.cascade_threshold
195
+
196
+ def _combine_results(
197
+ self,
198
+ pattern_result: JudgeResult,
199
+ llm_result: Optional[JudgeResult],
200
+ language: str,
201
+ ) -> JudgeResult:
202
+ """Combine results based on consensus mode."""
203
+ # If no LLM result, return pattern result
204
+ if llm_result is None:
205
+ return JudgeResult(
206
+ is_secure=pattern_result.is_secure,
207
+ security_score=pattern_result.security_score,
208
+ findings=pattern_result.findings,
209
+ judge_type=self.judge_type,
210
+ language=language,
211
+ pattern_result=pattern_result,
212
+ llm_result=None,
213
+ consensus_mode=self.consensus_mode,
214
+ )
215
+
216
+ # Combine based on mode
217
+ if self.consensus_mode == ConsensusMode.ANY:
218
+ return self._combine_any(pattern_result, llm_result, language)
219
+ elif self.consensus_mode == ConsensusMode.BOTH:
220
+ return self._combine_both(pattern_result, llm_result, language)
221
+ elif self.consensus_mode == ConsensusMode.WEIGHTED:
222
+ return self._combine_weighted(pattern_result, llm_result, language)
223
+ elif self.consensus_mode == ConsensusMode.CASCADE:
224
+ return self._combine_cascade(pattern_result, llm_result, language)
225
+ else:
226
+ raise ValueError(f"Unknown consensus mode: {self.consensus_mode}")
227
+
228
+ def _combine_any(
229
+ self,
230
+ pattern_result: JudgeResult,
231
+ llm_result: JudgeResult,
232
+ language: str,
233
+ ) -> JudgeResult:
234
+ """ANY mode: Include findings from either judge (union)."""
235
+ # Merge all findings
236
+ all_findings = list(pattern_result.findings) + list(llm_result.findings)
237
+
238
+ # Deduplicate by CWE and location
239
+ unique_findings = self._deduplicate_findings(all_findings)
240
+
241
+ # Insecure if either judge says insecure
242
+ is_secure = pattern_result.is_secure and llm_result.is_secure
243
+
244
+ # Score is minimum of both
245
+ security_score = min(
246
+ pattern_result.security_score, llm_result.security_score
247
+ )
248
+
249
+ return JudgeResult(
250
+ is_secure=is_secure,
251
+ security_score=security_score,
252
+ findings=unique_findings,
253
+ judge_type=self.judge_type,
254
+ language=language,
255
+ pattern_result=pattern_result,
256
+ llm_result=llm_result,
257
+ consensus_mode=self.consensus_mode,
258
+ )
259
+
260
+ def _combine_both(
261
+ self,
262
+ pattern_result: JudgeResult,
263
+ llm_result: JudgeResult,
264
+ language: str,
265
+ ) -> JudgeResult:
266
+ """BOTH mode: Only include findings both judges agree on (intersection)."""
267
+ # Find matching findings
268
+ agreed_findings = self._find_agreed_findings(
269
+ pattern_result.findings, llm_result.findings
270
+ )
271
+
272
+ return JudgeResult(
273
+ is_secure=self._is_secure(agreed_findings),
274
+ security_score=self._compute_score(agreed_findings),
275
+ findings=agreed_findings,
276
+ judge_type=self.judge_type,
277
+ language=language,
278
+ pattern_result=pattern_result,
279
+ llm_result=llm_result,
280
+ consensus_mode=self.consensus_mode,
281
+ )
282
+
283
+ def _combine_weighted(
284
+ self,
285
+ pattern_result: JudgeResult,
286
+ llm_result: JudgeResult,
287
+ language: str,
288
+ ) -> JudgeResult:
289
+ """WEIGHTED mode: Weighted combination of confidences."""
290
+ # Merge findings with weighted confidence
291
+ weighted_findings = []
292
+
293
+ # Group findings by CWE
294
+ pattern_by_cwe = self._group_by_cwe(pattern_result.findings)
295
+ llm_by_cwe = self._group_by_cwe(llm_result.findings)
296
+
297
+ all_cwes = set(pattern_by_cwe.keys()) | set(llm_by_cwe.keys())
298
+
299
+ for cwe in all_cwes:
300
+ pattern_findings = pattern_by_cwe.get(cwe, [])
301
+ llm_findings = llm_by_cwe.get(cwe, [])
302
+
303
+ if pattern_findings and llm_findings:
304
+ # Both found - combine confidence
305
+ pattern_conf = max(f.confidence for f in pattern_findings)
306
+ llm_conf = max(f.confidence for f in llm_findings)
307
+ combined_conf = (
308
+ self.pattern_weight * pattern_conf
309
+ + self.llm_weight * llm_conf
310
+ )
311
+
312
+ # Use LLM finding as base (better description/reasoning)
313
+ best_llm = max(llm_findings, key=lambda f: f.confidence)
314
+ combined = JudgeFinding(
315
+ cwe_id=best_llm.cwe_id,
316
+ vulnerability_type=best_llm.vulnerability_type,
317
+ description=best_llm.description,
318
+ severity=best_llm.severity,
319
+ confidence=combined_conf,
320
+ location=best_llm.location or (
321
+ pattern_findings[0].location if pattern_findings else None
322
+ ),
323
+ suggested_fix=best_llm.suggested_fix,
324
+ judge_type="dual",
325
+ reasoning=best_llm.reasoning,
326
+ )
327
+ weighted_findings.append(combined)
328
+
329
+ elif pattern_findings:
330
+ # Only pattern found - use pattern weight
331
+ for f in pattern_findings:
332
+ adjusted = JudgeFinding(
333
+ cwe_id=f.cwe_id,
334
+ vulnerability_type=f.vulnerability_type,
335
+ description=f.description,
336
+ severity=f.severity,
337
+ confidence=f.confidence * self.pattern_weight,
338
+ location=f.location,
339
+ suggested_fix=f.suggested_fix,
340
+ judge_type="pattern",
341
+ reasoning=f.reasoning,
342
+ )
343
+ weighted_findings.append(adjusted)
344
+
345
+ else:
346
+ # Only LLM found - use LLM weight
347
+ for f in llm_findings:
348
+ adjusted = JudgeFinding(
349
+ cwe_id=f.cwe_id,
350
+ vulnerability_type=f.vulnerability_type,
351
+ description=f.description,
352
+ severity=f.severity,
353
+ confidence=f.confidence * self.llm_weight,
354
+ location=f.location,
355
+ suggested_fix=f.suggested_fix,
356
+ judge_type="llm",
357
+ reasoning=f.reasoning,
358
+ )
359
+ weighted_findings.append(adjusted)
360
+
361
+ # Filter by confidence threshold
362
+ filtered = self._filter_findings(weighted_findings)
363
+
364
+ # Weighted score
365
+ weighted_score = (
366
+ self.pattern_weight * pattern_result.security_score
367
+ + self.llm_weight * llm_result.security_score
368
+ )
369
+
370
+ return JudgeResult(
371
+ is_secure=self._is_secure(filtered),
372
+ security_score=weighted_score,
373
+ findings=filtered,
374
+ judge_type=self.judge_type,
375
+ language=language,
376
+ pattern_result=pattern_result,
377
+ llm_result=llm_result,
378
+ consensus_mode=self.consensus_mode,
379
+ )
380
+
381
+ def _combine_cascade(
382
+ self,
383
+ pattern_result: JudgeResult,
384
+ llm_result: JudgeResult,
385
+ language: str,
386
+ ) -> JudgeResult:
387
+ """CASCADE mode: Pattern first, LLM validates/refines."""
388
+ # LLM result refines pattern result
389
+ # - Confirms or rejects pattern findings
390
+ # - May add new findings pattern missed
391
+
392
+ final_findings = []
393
+
394
+ # Check which pattern findings LLM confirms
395
+ pattern_by_cwe = self._group_by_cwe(pattern_result.findings)
396
+ llm_by_cwe = self._group_by_cwe(llm_result.findings)
397
+
398
+ for cwe, pattern_findings in pattern_by_cwe.items():
399
+ if cwe in llm_by_cwe:
400
+ # LLM confirms - boost confidence
401
+ for f in pattern_findings:
402
+ confirmed = JudgeFinding(
403
+ cwe_id=f.cwe_id,
404
+ vulnerability_type=f.vulnerability_type,
405
+ description=f.description,
406
+ severity=f.severity,
407
+ confidence=min(1.0, f.confidence * 1.3), # Boost
408
+ location=f.location,
409
+ suggested_fix=llm_by_cwe[cwe][0].suggested_fix or f.suggested_fix,
410
+ judge_type="dual",
411
+ reasoning=llm_by_cwe[cwe][0].reasoning,
412
+ )
413
+ final_findings.append(confirmed)
414
+ else:
415
+ # LLM doesn't confirm - reduce confidence
416
+ for f in pattern_findings:
417
+ unconfirmed = JudgeFinding(
418
+ cwe_id=f.cwe_id,
419
+ vulnerability_type=f.vulnerability_type,
420
+ description=f.description,
421
+ severity=f.severity,
422
+ confidence=f.confidence * 0.5, # Reduce
423
+ location=f.location,
424
+ suggested_fix=f.suggested_fix,
425
+ judge_type="pattern",
426
+ reasoning="LLM did not confirm this finding",
427
+ )
428
+ final_findings.append(unconfirmed)
429
+
430
+ # Add LLM-only findings (pattern false negatives)
431
+ for cwe, llm_findings in llm_by_cwe.items():
432
+ if cwe not in pattern_by_cwe:
433
+ for f in llm_findings:
434
+ final_findings.append(f)
435
+
436
+ # Filter
437
+ filtered = self._filter_findings(final_findings)
438
+
439
+ return JudgeResult(
440
+ is_secure=self._is_secure(filtered),
441
+ security_score=self._compute_score(filtered),
442
+ findings=filtered,
443
+ judge_type=self.judge_type,
444
+ language=language,
445
+ pattern_result=pattern_result,
446
+ llm_result=llm_result,
447
+ consensus_mode=self.consensus_mode,
448
+ )
449
+
450
+ def _deduplicate_findings(
451
+ self,
452
+ findings: List[JudgeFinding],
453
+ ) -> List[JudgeFinding]:
454
+ """Remove duplicate findings, keeping highest confidence."""
455
+ # Group by (CWE, line)
456
+ by_key: Dict[Tuple[str, Optional[int]], JudgeFinding] = {}
457
+
458
+ for f in findings:
459
+ line = f.location.line if f.location else None
460
+ key = (f.cwe_id, line)
461
+
462
+ if key not in by_key or f.confidence > by_key[key].confidence:
463
+ by_key[key] = f
464
+
465
+ return list(by_key.values())
466
+
467
+ def _find_agreed_findings(
468
+ self,
469
+ pattern_findings: List[JudgeFinding],
470
+ llm_findings: List[JudgeFinding],
471
+ ) -> List[JudgeFinding]:
472
+ """Find findings that both judges agree on."""
473
+ pattern_cwes = {f.cwe_id for f in pattern_findings}
474
+ llm_cwes = {f.cwe_id for f in llm_findings}
475
+ agreed_cwes = pattern_cwes & llm_cwes
476
+
477
+ # Return LLM findings for agreed CWEs (better descriptions)
478
+ return [f for f in llm_findings if f.cwe_id in agreed_cwes]
479
+
480
+ def _group_by_cwe(
481
+ self,
482
+ findings: List[JudgeFinding],
483
+ ) -> Dict[str, List[JudgeFinding]]:
484
+ """Group findings by CWE ID."""
485
+ by_cwe: Dict[str, List[JudgeFinding]] = {}
486
+ for f in findings:
487
+ if f.cwe_id not in by_cwe:
488
+ by_cwe[f.cwe_id] = []
489
+ by_cwe[f.cwe_id].append(f)
490
+ return by_cwe
491
+
492
+ @classmethod
493
+ def pattern_only(cls) -> "DualJudge":
494
+ """Factory for pattern-only mode (fast, no API calls)."""
495
+ return cls(
496
+ pattern_judge=PatternJudge(),
497
+ llm_judge=None,
498
+ )
499
+
500
+ @classmethod
501
+ def high_recall(cls, llm_model: str = "gpt-4") -> "DualJudge":
502
+ """Factory for high recall mode (catches more vulnerabilities)."""
503
+ return cls(
504
+ pattern_judge=PatternJudge.with_strict_rules(),
505
+ llm_judge=LLMJudge(model=llm_model),
506
+ consensus_mode=ConsensusMode.ANY,
507
+ )
508
+
509
+ @classmethod
510
+ def high_precision(cls, llm_model: str = "gpt-4") -> "DualJudge":
511
+ """Factory for high precision mode (fewer false positives)."""
512
+ return cls(
513
+ pattern_judge=PatternJudge.with_high_precision(),
514
+ llm_judge=LLMJudge(model=llm_model),
515
+ consensus_mode=ConsensusMode.BOTH,
516
+ )
517
+
518
+ @classmethod
519
+ def balanced(cls, llm_model: str = "gpt-4") -> "DualJudge":
520
+ """Factory for balanced mode (default)."""
521
+ return cls(
522
+ pattern_judge=PatternJudge(),
523
+ llm_judge=LLMJudge(model=llm_model),
524
+ consensus_mode=ConsensusMode.WEIGHTED,
525
+ )
526
+
527
+ @classmethod
528
+ def efficient(cls, llm_model: str = "gpt-3.5-turbo") -> "DualJudge":
529
+ """Factory for efficient mode (pattern first, LLM only when needed)."""
530
+ return cls(
531
+ pattern_judge=PatternJudge(),
532
+ llm_judge=LLMJudge(model=llm_model),
533
+ consensus_mode=ConsensusMode.CASCADE,
534
+ )