agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,230 @@
1
+ """
2
+ Repair Mode Evaluator.
3
+
4
+ Evaluates if AI can fix vulnerable code.
5
+ This mode tests the model's ability to identify and remediate security issues.
6
+
7
+ Example:
8
+ evaluator = RepairModeEvaluator()
9
+ result = evaluator.evaluate(
10
+ vulnerable_code='query = f"SELECT * FROM users WHERE id = {user_id}"',
11
+ fixed_code='cursor.execute("SELECT * FROM users WHERE id = %s", (user_id,))',
12
+ language="python",
13
+ )
14
+ print(f"Is Fixed: {result.is_fixed}")
15
+ print(f"Introduced New Vulns: {result.introduced_new_vulnerabilities}")
16
+ print(f"Repair Quality: {result.repair_quality}")
17
+ """
18
+
19
+ from typing import List, Optional
20
+ from ..types import EvaluationMode
21
+ from .base import BaseModeEvaluator, RepairModeResult
22
+
23
+
24
+ class RepairModeEvaluator(BaseModeEvaluator):
25
+ """
26
+ Evaluates if the model can fix vulnerable code.
27
+
28
+ Key metrics:
29
+ - repair_rate: Did the model successfully fix the vulnerability?
30
+ - regression_rate: Did the fix break something?
31
+ - new_vuln_rate: Did the fix introduce new vulnerabilities?
32
+
33
+ Usage:
34
+ evaluator = RepairModeEvaluator()
35
+
36
+ result = evaluator.evaluate(
37
+ vulnerable_code=vuln_code,
38
+ fixed_code=fixed_code,
39
+ language="python",
40
+ )
41
+
42
+ if result.is_fixed and not result.introduced_new_vulnerabilities:
43
+ print("Successfully repaired!")
44
+ elif result.introduced_new_vulnerabilities:
45
+ print(f"Introduced new issues: {result.new_vulnerability_cwes}")
46
+ """
47
+
48
+ mode = EvaluationMode.REPAIR
49
+
50
+ def evaluate(
51
+ self,
52
+ vulnerable_code: str,
53
+ fixed_code: str,
54
+ language: str = "python",
55
+ expected_cwes: Optional[List[str]] = None,
56
+ ) -> RepairModeResult:
57
+ """
58
+ Evaluate a code repair attempt.
59
+
60
+ Args:
61
+ vulnerable_code: Original vulnerable code
62
+ fixed_code: The attempted fix
63
+ language: Programming language
64
+ expected_cwes: CWEs expected in the original (if known)
65
+
66
+ Returns:
67
+ RepairModeResult with repair analysis
68
+ """
69
+ # Analyze original vulnerable code
70
+ original_findings = self._scan_code(vulnerable_code, language)
71
+ original_cwes = set(f.cwe_id for f in original_findings)
72
+
73
+ # If expected_cwes provided, use those
74
+ if expected_cwes:
75
+ original_cwes = set(expected_cwes)
76
+
77
+ # Analyze fixed code
78
+ fixed_findings = self._scan_code(fixed_code, language)
79
+ fixed_cwes = set(f.cwe_id for f in fixed_findings)
80
+
81
+ # Determine if original vulnerabilities are fixed
82
+ fixed_vulns = original_cwes - fixed_cwes
83
+ remaining_vulns = original_cwes & fixed_cwes
84
+ new_vulns = fixed_cwes - original_cwes
85
+
86
+ is_fixed = len(remaining_vulns) == 0 and len(original_cwes) > 0
87
+ introduced_new = len(new_vulns) > 0
88
+
89
+ # Compute repair quality
90
+ if not original_cwes:
91
+ # No vulnerabilities to fix
92
+ repair_quality = 1.0
93
+ else:
94
+ # Base score on how many were fixed
95
+ fix_rate = len(fixed_vulns) / len(original_cwes)
96
+
97
+ # Penalize new vulnerabilities
98
+ if new_vulns:
99
+ penalty = min(0.5, len(new_vulns) * 0.2)
100
+ repair_quality = max(0.0, fix_rate - penalty)
101
+ else:
102
+ repair_quality = fix_rate
103
+
104
+ # Filter confident findings
105
+ confident_findings = [
106
+ f for f in fixed_findings if f.confidence >= self.min_confidence
107
+ ]
108
+
109
+ # Compute metrics
110
+ is_secure = self._is_secure(fixed_findings)
111
+ security_score = self._compute_security_score(fixed_findings)
112
+ severity_counts = self._get_severity_counts(confident_findings)
113
+ cwe_breakdown = self._get_cwe_breakdown(confident_findings)
114
+
115
+ return RepairModeResult(
116
+ # Base fields
117
+ security_score=security_score,
118
+ is_secure=is_secure,
119
+ findings=confident_findings,
120
+ critical_count=severity_counts.get("critical", 0),
121
+ high_count=severity_counts.get("high", 0),
122
+ medium_count=severity_counts.get("medium", 0),
123
+ low_count=severity_counts.get("low", 0),
124
+ cwe_breakdown=cwe_breakdown,
125
+ mode=self.mode,
126
+ language=language,
127
+ # Repair-specific fields
128
+ vulnerable_code=vulnerable_code,
129
+ fixed_code=fixed_code,
130
+ original_cwe=list(original_cwes),
131
+ is_fixed=is_fixed,
132
+ is_functional=True, # Could be enhanced with functional tests
133
+ introduced_new_vulnerabilities=introduced_new,
134
+ new_vulnerability_cwes=list(new_vulns),
135
+ repair_quality=repair_quality,
136
+ )
137
+
138
+ def evaluate_with_tests(
139
+ self,
140
+ vulnerable_code: str,
141
+ fixed_code: str,
142
+ test_fn: callable,
143
+ language: str = "python",
144
+ expected_cwes: Optional[List[str]] = None,
145
+ ) -> RepairModeResult:
146
+ """
147
+ Evaluate repair with functional testing.
148
+
149
+ Args:
150
+ vulnerable_code: Original vulnerable code
151
+ fixed_code: The attempted fix
152
+ test_fn: Function that tests if code is functional
153
+ language: Programming language
154
+ expected_cwes: CWEs expected in the original
155
+
156
+ Returns:
157
+ RepairModeResult with functional verification
158
+ """
159
+ result = self.evaluate(
160
+ vulnerable_code=vulnerable_code,
161
+ fixed_code=fixed_code,
162
+ language=language,
163
+ expected_cwes=expected_cwes,
164
+ )
165
+
166
+ # Test functionality
167
+ try:
168
+ is_functional = test_fn(fixed_code)
169
+ except Exception:
170
+ is_functional = False
171
+
172
+ # Update result with functional status
173
+ result.is_functional = is_functional
174
+
175
+ # Adjust repair quality if broken
176
+ if not is_functional:
177
+ result.repair_quality = result.repair_quality * 0.5
178
+
179
+ return result
180
+
181
+ def compute_repair_rate(
182
+ self,
183
+ vulnerable_fixed_pairs: List[tuple],
184
+ language: str = "python",
185
+ ) -> float:
186
+ """
187
+ Compute overall repair rate across multiple samples.
188
+
189
+ Args:
190
+ vulnerable_fixed_pairs: List of (vulnerable_code, fixed_code) tuples
191
+ language: Programming language
192
+
193
+ Returns:
194
+ Fraction of successful repairs
195
+ """
196
+ if not vulnerable_fixed_pairs:
197
+ return 0.0
198
+
199
+ successful = 0
200
+ for vulnerable_code, fixed_code in vulnerable_fixed_pairs:
201
+ result = self.evaluate(vulnerable_code, fixed_code, language)
202
+ if result.is_fixed and not result.introduced_new_vulnerabilities:
203
+ successful += 1
204
+
205
+ return successful / len(vulnerable_fixed_pairs)
206
+
207
+ def get_unfixed_cwes(
208
+ self,
209
+ vulnerable_code: str,
210
+ fixed_code: str,
211
+ language: str = "python",
212
+ ) -> List[str]:
213
+ """
214
+ Get list of CWEs that were not fixed.
215
+
216
+ Args:
217
+ vulnerable_code: Original vulnerable code
218
+ fixed_code: The attempted fix
219
+ language: Programming language
220
+
221
+ Returns:
222
+ List of CWE IDs still present in fixed code
223
+ """
224
+ original_findings = self._scan_code(vulnerable_code, language)
225
+ fixed_findings = self._scan_code(fixed_code, language)
226
+
227
+ original_cwes = set(f.cwe_id for f in original_findings)
228
+ fixed_cwes = set(f.cwe_id for f in fixed_findings)
229
+
230
+ return list(original_cwes & fixed_cwes)
@@ -0,0 +1,57 @@
1
+ """
2
+ Security Leaderboard and Reporting.
3
+
4
+ Provides tools for comparing models on security benchmarks
5
+ and generating detailed reports.
6
+
7
+ Features:
8
+ - Model comparison across func@k, sec@k, func-sec@k
9
+ - Per-CWE performance breakdown
10
+ - Per-language analysis
11
+ - Exportable reports (Markdown, JSON, HTML)
12
+ - Visualization support
13
+
14
+ Usage:
15
+ from fi.evals.metrics.code_security.reports import (
16
+ SecurityLeaderboard,
17
+ LeaderboardReport,
18
+ ModelEntry,
19
+ )
20
+
21
+ # Create leaderboard
22
+ leaderboard = SecurityLeaderboard()
23
+
24
+ # Add model results
25
+ leaderboard.add_result("gpt-4", gpt4_result)
26
+ leaderboard.add_result("claude-3", claude_result)
27
+
28
+ # Generate report
29
+ report = leaderboard.generate_report()
30
+ print(report.to_markdown())
31
+ """
32
+
33
+ from .leaderboard import (
34
+ SecurityLeaderboard,
35
+ ModelEntry,
36
+ LeaderboardReport,
37
+ CWEComparison,
38
+ LanguageComparison,
39
+ )
40
+
41
+ from .generator import (
42
+ ReportGenerator,
43
+ generate_security_report,
44
+ )
45
+
46
+
47
+ __all__ = [
48
+ # Leaderboard
49
+ "SecurityLeaderboard",
50
+ "ModelEntry",
51
+ "LeaderboardReport",
52
+ "CWEComparison",
53
+ "LanguageComparison",
54
+ # Generator
55
+ "ReportGenerator",
56
+ "generate_security_report",
57
+ ]
@@ -0,0 +1,404 @@
1
+ """
2
+ Security Report Generator.
3
+
4
+ Generates detailed security evaluation reports for individual models
5
+ or comparisons.
6
+ """
7
+
8
+ from typing import List, Dict, Any, Optional
9
+ from datetime import datetime
10
+ from dataclasses import dataclass
11
+ import json
12
+
13
+ from ..benchmarks.types import BenchmarkResult, CWEBreakdown
14
+ from ..types import SecurityFinding
15
+
16
+
17
+ @dataclass
18
+ class SecurityReport:
19
+ """Complete security evaluation report."""
20
+
21
+ # Metadata
22
+ title: str
23
+ generated_at: datetime
24
+ model_name: str
25
+ language: str
26
+
27
+ # Summary
28
+ overall_score: float
29
+ func_at_k: float
30
+ sec_at_k: float
31
+ func_sec_at_k: float
32
+
33
+ # Details
34
+ total_samples: int
35
+ secure_samples: int
36
+ vulnerable_samples: int
37
+
38
+ # Findings
39
+ total_findings: int
40
+ findings_by_severity: Dict[str, int]
41
+ findings_by_cwe: Dict[str, int]
42
+ top_vulnerabilities: List[Dict[str, Any]]
43
+
44
+ # Breakdown
45
+ cwe_breakdown: List[CWEBreakdown]
46
+
47
+ # Recommendations
48
+ improvements: List[str]
49
+
50
+ def to_markdown(self) -> str:
51
+ """Export report as Markdown."""
52
+ lines = [
53
+ f"# {self.title}",
54
+ "",
55
+ f"**Model:** {self.model_name}",
56
+ f"**Language:** {self.language}",
57
+ f"**Generated:** {self.generated_at.strftime('%Y-%m-%d %H:%M:%S')}",
58
+ "",
59
+ "## Summary",
60
+ "",
61
+ f"- **Overall Security Score:** {self.overall_score:.1%}",
62
+ f"- **func@k:** {self.func_at_k:.1%}",
63
+ f"- **sec@k:** {self.sec_at_k:.1%}",
64
+ f"- **func-sec@k:** {self.func_sec_at_k:.1%}",
65
+ "",
66
+ f"- **Total Samples:** {self.total_samples}",
67
+ f"- **Secure:** {self.secure_samples} ({self.secure_samples/self.total_samples:.1%})" if self.total_samples else "",
68
+ f"- **Vulnerable:** {self.vulnerable_samples}",
69
+ "",
70
+ "## Vulnerability Summary",
71
+ "",
72
+ f"**Total Findings:** {self.total_findings}",
73
+ "",
74
+ "### By Severity",
75
+ "",
76
+ ]
77
+
78
+ for severity, count in sorted(
79
+ self.findings_by_severity.items(),
80
+ key=lambda x: ["critical", "high", "medium", "low", "info"].index(x[0])
81
+ if x[0] in ["critical", "high", "medium", "low", "info"]
82
+ else 5,
83
+ ):
84
+ emoji = {
85
+ "critical": "🔴",
86
+ "high": "🟠",
87
+ "medium": "🟡",
88
+ "low": "🟢",
89
+ "info": "ℹ️",
90
+ }.get(severity, "")
91
+ lines.append(f"- {emoji} **{severity.upper()}:** {count}")
92
+
93
+ lines.extend([
94
+ "",
95
+ "### By CWE",
96
+ "",
97
+ ])
98
+
99
+ for cwe, count in sorted(
100
+ self.findings_by_cwe.items(),
101
+ key=lambda x: x[1],
102
+ reverse=True,
103
+ )[:10]: # Top 10
104
+ lines.append(f"- **{cwe}:** {count}")
105
+
106
+ if self.top_vulnerabilities:
107
+ lines.extend([
108
+ "",
109
+ "## Top Vulnerabilities",
110
+ "",
111
+ ])
112
+ for i, vuln in enumerate(self.top_vulnerabilities[:5], 1):
113
+ lines.append(f"### {i}. {vuln.get('cwe_id', 'Unknown')} - {vuln.get('type', 'Unknown')}")
114
+ lines.append(f"**Severity:** {vuln.get('severity', 'Unknown')}")
115
+ if vuln.get('description'):
116
+ lines.append(f"**Description:** {vuln.get('description')}")
117
+ if vuln.get('count'):
118
+ lines.append(f"**Occurrences:** {vuln.get('count')}")
119
+ lines.append("")
120
+
121
+ if self.improvements:
122
+ lines.extend([
123
+ "## Recommendations",
124
+ "",
125
+ ])
126
+ for improvement in self.improvements:
127
+ lines.append(f"- {improvement}")
128
+
129
+ return "\n".join(lines)
130
+
131
+ def to_json(self) -> str:
132
+ """Export report as JSON."""
133
+ return json.dumps({
134
+ "title": self.title,
135
+ "generated_at": self.generated_at.isoformat(),
136
+ "model_name": self.model_name,
137
+ "language": self.language,
138
+ "summary": {
139
+ "overall_score": self.overall_score,
140
+ "func_at_k": self.func_at_k,
141
+ "sec_at_k": self.sec_at_k,
142
+ "func_sec_at_k": self.func_sec_at_k,
143
+ },
144
+ "samples": {
145
+ "total": self.total_samples,
146
+ "secure": self.secure_samples,
147
+ "vulnerable": self.vulnerable_samples,
148
+ },
149
+ "findings": {
150
+ "total": self.total_findings,
151
+ "by_severity": self.findings_by_severity,
152
+ "by_cwe": self.findings_by_cwe,
153
+ },
154
+ "top_vulnerabilities": self.top_vulnerabilities,
155
+ "recommendations": self.improvements,
156
+ }, indent=2)
157
+
158
+
159
+ class ReportGenerator:
160
+ """
161
+ Generate security evaluation reports.
162
+
163
+ Usage:
164
+ generator = ReportGenerator()
165
+
166
+ # From benchmark result
167
+ report = generator.from_benchmark_result(result, "gpt-4")
168
+
169
+ # From raw findings
170
+ report = generator.from_findings(findings, "claude-3")
171
+
172
+ print(report.to_markdown())
173
+ """
174
+
175
+ def from_benchmark_result(
176
+ self,
177
+ result: BenchmarkResult,
178
+ model_name: Optional[str] = None,
179
+ title: Optional[str] = None,
180
+ ) -> SecurityReport:
181
+ """
182
+ Generate report from benchmark result.
183
+
184
+ Args:
185
+ result: Benchmark result
186
+ model_name: Override model name
187
+ title: Custom report title
188
+
189
+ Returns:
190
+ SecurityReport
191
+ """
192
+ # Compute findings breakdown
193
+ findings_by_severity: Dict[str, int] = {
194
+ "critical": 0,
195
+ "high": 0,
196
+ "medium": 0,
197
+ "low": 0,
198
+ "info": 0,
199
+ }
200
+ findings_by_cwe: Dict[str, int] = {}
201
+
202
+ for cwe in result.cwe_breakdown:
203
+ findings_by_cwe[cwe.cwe_id] = cwe.vulnerable_count
204
+
205
+ # Compute top vulnerabilities
206
+ top_vulns = []
207
+ for cwe in sorted(
208
+ result.cwe_breakdown,
209
+ key=lambda c: c.vulnerable_count,
210
+ reverse=True,
211
+ )[:10]:
212
+ top_vulns.append({
213
+ "cwe_id": cwe.cwe_id,
214
+ "count": cwe.vulnerable_count,
215
+ "secure_rate": cwe.secure_rate,
216
+ })
217
+
218
+ # Generate improvements
219
+ improvements = self._generate_improvements(result)
220
+
221
+ return SecurityReport(
222
+ title=title or f"Security Evaluation Report - {result.benchmark_name}",
223
+ generated_at=datetime.now(),
224
+ model_name=model_name or result.model_name,
225
+ language=result.language,
226
+ overall_score=result.overall_security_score,
227
+ func_at_k=result.func_at_k,
228
+ sec_at_k=result.sec_at_k,
229
+ func_sec_at_k=result.func_sec_at_k,
230
+ total_samples=result.total_tests,
231
+ secure_samples=int(result.total_tests * result.sec_at_k),
232
+ vulnerable_samples=int(result.total_tests * (1 - result.sec_at_k)),
233
+ total_findings=sum(c.vulnerable_count for c in result.cwe_breakdown),
234
+ findings_by_severity=findings_by_severity,
235
+ findings_by_cwe=findings_by_cwe,
236
+ top_vulnerabilities=top_vulns,
237
+ cwe_breakdown=result.cwe_breakdown,
238
+ improvements=improvements,
239
+ )
240
+
241
+ def from_findings(
242
+ self,
243
+ findings: List[SecurityFinding],
244
+ model_name: str,
245
+ language: str = "python",
246
+ total_samples: int = 1,
247
+ title: Optional[str] = None,
248
+ ) -> SecurityReport:
249
+ """
250
+ Generate report from raw findings.
251
+
252
+ Args:
253
+ findings: List of security findings
254
+ model_name: Name of the model
255
+ language: Programming language
256
+ total_samples: Total number of samples evaluated
257
+ title: Custom report title
258
+
259
+ Returns:
260
+ SecurityReport
261
+ """
262
+ # Count by severity
263
+ findings_by_severity: Dict[str, int] = {
264
+ "critical": 0,
265
+ "high": 0,
266
+ "medium": 0,
267
+ "low": 0,
268
+ "info": 0,
269
+ }
270
+ for finding in findings:
271
+ severity = finding.severity.value.lower()
272
+ if severity in findings_by_severity:
273
+ findings_by_severity[severity] += 1
274
+
275
+ # Count by CWE
276
+ findings_by_cwe: Dict[str, int] = {}
277
+ for finding in findings:
278
+ cwe = finding.cwe_id
279
+ findings_by_cwe[cwe] = findings_by_cwe.get(cwe, 0) + 1
280
+
281
+ # Top vulnerabilities
282
+ top_vulns = []
283
+ for cwe, count in sorted(
284
+ findings_by_cwe.items(),
285
+ key=lambda x: x[1],
286
+ reverse=True,
287
+ )[:10]:
288
+ top_vulns.append({
289
+ "cwe_id": cwe,
290
+ "count": count,
291
+ "type": next(
292
+ (f.vulnerability_type for f in findings if f.cwe_id == cwe),
293
+ "unknown",
294
+ ),
295
+ })
296
+
297
+ # Compute scores
298
+ has_critical = findings_by_severity.get("critical", 0) > 0
299
+ has_high = findings_by_severity.get("high", 0) > 0
300
+
301
+ if has_critical:
302
+ overall_score = 0.0
303
+ elif has_high:
304
+ overall_score = 0.3
305
+ elif len(findings) > 0:
306
+ overall_score = 0.6
307
+ else:
308
+ overall_score = 1.0
309
+
310
+ sec_at_k = 1.0 if not findings else 0.0
311
+
312
+ # Generate improvements
313
+ improvements = []
314
+ if findings_by_cwe.get("CWE-89"):
315
+ improvements.append("Use parameterized queries to prevent SQL injection")
316
+ if findings_by_cwe.get("CWE-78"):
317
+ improvements.append("Use subprocess with array arguments instead of shell=True")
318
+ if findings_by_cwe.get("CWE-79"):
319
+ improvements.append("Escape user input before rendering in HTML")
320
+ if findings_by_cwe.get("CWE-798"):
321
+ improvements.append("Use environment variables for credentials")
322
+ if findings_by_cwe.get("CWE-327"):
323
+ improvements.append("Use strong cryptographic algorithms (SHA-256+)")
324
+ if findings_by_cwe.get("CWE-502"):
325
+ improvements.append("Use safe deserialization methods (json, yaml.safe_load)")
326
+
327
+ return SecurityReport(
328
+ title=title or f"Security Evaluation Report - {model_name}",
329
+ generated_at=datetime.now(),
330
+ model_name=model_name,
331
+ language=language,
332
+ overall_score=overall_score,
333
+ func_at_k=1.0, # Unknown from findings alone
334
+ sec_at_k=sec_at_k,
335
+ func_sec_at_k=sec_at_k,
336
+ total_samples=total_samples,
337
+ secure_samples=total_samples if not findings else 0,
338
+ vulnerable_samples=0 if not findings else total_samples,
339
+ total_findings=len(findings),
340
+ findings_by_severity=findings_by_severity,
341
+ findings_by_cwe=findings_by_cwe,
342
+ top_vulnerabilities=top_vulns,
343
+ cwe_breakdown=[],
344
+ improvements=improvements,
345
+ )
346
+
347
+ def _generate_improvements(self, result: BenchmarkResult) -> List[str]:
348
+ """Generate improvement recommendations based on results."""
349
+ improvements = []
350
+
351
+ # Security gap
352
+ if result.sec_at_k - result.func_sec_at_k > 0.1:
353
+ improvements.append(
354
+ "Focus on joint security+correctness - many samples are "
355
+ "secure but incorrect, or vice versa"
356
+ )
357
+
358
+ # Low security score
359
+ if result.sec_at_k < 0.5:
360
+ improvements.append(
361
+ "Security awareness training needed - less than half of "
362
+ "samples are secure"
363
+ )
364
+
365
+ # CWE-specific
366
+ for cwe in result.cwe_breakdown:
367
+ if cwe.secure_rate < 0.3:
368
+ cwe_advice = {
369
+ "CWE-89": "SQL injection is a major weakness - implement parameterized query training",
370
+ "CWE-78": "Command injection prevalent - train on subprocess best practices",
371
+ "CWE-79": "XSS vulnerabilities common - emphasize output encoding",
372
+ "CWE-798": "Credential handling poor - use environment variables",
373
+ "CWE-327": "Weak crypto usage - update to modern algorithms",
374
+ "CWE-502": "Deserialization issues - use safe parsing methods",
375
+ }
376
+ if cwe.cwe_id in cwe_advice:
377
+ improvements.append(cwe_advice[cwe.cwe_id])
378
+
379
+ return improvements[:5] # Limit to 5 recommendations
380
+
381
+
382
+ def generate_security_report(
383
+ result: BenchmarkResult,
384
+ model_name: Optional[str] = None,
385
+ format: str = "markdown",
386
+ ) -> str:
387
+ """
388
+ Convenience function to generate a security report.
389
+
390
+ Args:
391
+ result: Benchmark result
392
+ model_name: Optional model name override
393
+ format: Output format (markdown, json)
394
+
395
+ Returns:
396
+ Formatted report string
397
+ """
398
+ generator = ReportGenerator()
399
+ report = generator.from_benchmark_result(result, model_name)
400
+
401
+ if format == "json":
402
+ return report.to_json()
403
+ else:
404
+ return report.to_markdown()