agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,574 @@
1
+ """
2
+ Eval Delegate Scanner for Guardrails.
3
+
4
+ Delegates scanning to existing evaluation templates for overlapping functionality:
5
+ - PII Detection → Template 14, 22 (PII, DataPrivacyCompliance)
6
+ - Toxicity → Template 15 (Toxicity)
7
+ - Prompt Injection → Template 18 (PromptInjection)
8
+ - Bias Detection → Template 69, 77-79 (BiasDetection, NoRacialBias, etc.)
9
+ - Content Safety → Template 93 (ContentSafety)
10
+ - NSFW/Sexist → Template 17, 20 (Sexist, SafeForWorkText)
11
+
12
+ This scanner bridges the guardrails system with the evaluation framework,
13
+ allowing you to use battle-tested LLM-based evaluations as safety scanners.
14
+ """
15
+
16
+ import time
17
+ from enum import Enum
18
+ from typing import Any, Dict, List, Optional
19
+ from dataclasses import dataclass, field
20
+
21
+ from fi.evals.guardrails.scanners.base import (
22
+ BaseScanner,
23
+ ScanResult,
24
+ ScanMatch,
25
+ ScannerAction,
26
+ register_scanner,
27
+ )
28
+
29
+
30
+ class EvalCategory(str, Enum):
31
+ """Categories of evals that can be delegated to."""
32
+ PII = "pii"
33
+ TOXICITY = "toxicity"
34
+ PROMPT_INJECTION = "prompt_injection"
35
+ BIAS = "bias"
36
+ RACIAL_BIAS = "racial_bias"
37
+ GENDER_BIAS = "gender_bias"
38
+ AGE_BIAS = "age_bias"
39
+ CONTENT_SAFETY = "content_safety"
40
+ NSFW = "nsfw"
41
+ SEXIST = "sexist"
42
+
43
+
44
+ # Mapping from category to eval template info
45
+ EVAL_TEMPLATE_MAP: Dict[EvalCategory, Dict[str, Any]] = {
46
+ EvalCategory.PII: {
47
+ "eval_id": "14",
48
+ "eval_name": "pii",
49
+ "description": "Detects personally identifiable information",
50
+ "threshold": 0.5, # Score above this = PII detected
51
+ "invert": True, # High score = bad (PII found)
52
+ },
53
+ EvalCategory.TOXICITY: {
54
+ "eval_id": "15",
55
+ "eval_name": "toxicity",
56
+ "description": "Detects toxic or harmful content",
57
+ "threshold": 0.5,
58
+ "invert": True,
59
+ },
60
+ EvalCategory.PROMPT_INJECTION: {
61
+ "eval_id": "18",
62
+ "eval_name": "prompt_injection",
63
+ "description": "Detects prompt injection attempts",
64
+ "threshold": 0.5,
65
+ "invert": True,
66
+ },
67
+ EvalCategory.BIAS: {
68
+ "eval_id": "69",
69
+ "eval_name": "bias_detection",
70
+ "description": "Detects various forms of bias",
71
+ "threshold": 0.5,
72
+ "invert": True,
73
+ },
74
+ EvalCategory.RACIAL_BIAS: {
75
+ "eval_id": "77",
76
+ "eval_name": "no_racial_bias",
77
+ "description": "Detects racial bias",
78
+ "threshold": 0.5,
79
+ "invert": False, # High score = good (no bias)
80
+ },
81
+ EvalCategory.GENDER_BIAS: {
82
+ "eval_id": "78",
83
+ "eval_name": "no_gender_bias",
84
+ "description": "Detects gender bias",
85
+ "threshold": 0.5,
86
+ "invert": False,
87
+ },
88
+ EvalCategory.AGE_BIAS: {
89
+ "eval_id": "79",
90
+ "eval_name": "no_age_bias",
91
+ "description": "Detects age bias",
92
+ "threshold": 0.5,
93
+ "invert": False,
94
+ },
95
+ EvalCategory.CONTENT_SAFETY: {
96
+ "eval_id": "93",
97
+ "eval_name": "content_safety_violation",
98
+ "description": "Detects content safety violations",
99
+ "threshold": 0.5,
100
+ "invert": True,
101
+ },
102
+ EvalCategory.NSFW: {
103
+ "eval_id": "20",
104
+ "eval_name": "safe_for_work_text",
105
+ "description": "Detects NSFW content",
106
+ "threshold": 0.5,
107
+ "invert": False, # High score = safe
108
+ },
109
+ EvalCategory.SEXIST: {
110
+ "eval_id": "17",
111
+ "eval_name": "sexist",
112
+ "description": "Detects sexist content",
113
+ "threshold": 0.5,
114
+ "invert": True,
115
+ },
116
+ }
117
+
118
+
119
+ @dataclass
120
+ class EvalDelegateConfig:
121
+ """Configuration for EvalDelegateScanner."""
122
+ # Categories to check
123
+ categories: List[EvalCategory] = field(default_factory=lambda: [EvalCategory.TOXICITY])
124
+
125
+ # Thresholds per category (overrides defaults)
126
+ thresholds: Dict[EvalCategory, float] = field(default_factory=dict)
127
+
128
+ # Use local evaluator if available (faster, no API calls)
129
+ prefer_local: bool = True
130
+
131
+ # API key for cloud evaluation (if not using local)
132
+ api_key: Optional[str] = None
133
+
134
+ # Timeout for evaluation calls (seconds)
135
+ timeout: int = 30
136
+
137
+ # Whether to run categories in parallel
138
+ parallel: bool = True
139
+
140
+ # Aggregation mode: "any" (fail if any category fails) or "all" (fail only if all fail)
141
+ aggregation: str = "any"
142
+
143
+
144
+ @register_scanner("eval_delegate")
145
+ class EvalDelegateScanner(BaseScanner):
146
+ """
147
+ Scanner that delegates detection to existing evaluation templates.
148
+
149
+ This scanner bridges guardrails with the evaluation framework, using
150
+ proven LLM-based evaluations for safety detection.
151
+
152
+ Supported Categories:
153
+ - pii: Detect personally identifiable information
154
+ - toxicity: Detect toxic or harmful content
155
+ - prompt_injection: Detect prompt injection attempts
156
+ - bias: Detect various forms of bias
157
+ - racial_bias, gender_bias, age_bias: Specific bias detection
158
+ - content_safety: Detect content safety violations
159
+ - nsfw: Detect not-safe-for-work content
160
+ - sexist: Detect sexist content
161
+
162
+ Usage:
163
+ # Single category
164
+ scanner = EvalDelegateScanner(categories=[EvalCategory.TOXICITY])
165
+ result = scanner.scan("Your content here")
166
+
167
+ # Multiple categories
168
+ scanner = EvalDelegateScanner(
169
+ categories=[EvalCategory.TOXICITY, EvalCategory.PII, EvalCategory.BIAS]
170
+ )
171
+
172
+ # With custom thresholds
173
+ scanner = EvalDelegateScanner(
174
+ categories=[EvalCategory.TOXICITY],
175
+ thresholds={EvalCategory.TOXICITY: 0.7}
176
+ )
177
+
178
+ # Factory methods
179
+ scanner = EvalDelegateScanner.for_toxicity()
180
+ scanner = EvalDelegateScanner.for_pii()
181
+ scanner = EvalDelegateScanner.for_safety() # Multiple categories
182
+
183
+ Note:
184
+ This scanner requires either:
185
+ - API key for cloud-based evaluation (more accurate, slower)
186
+ - Local LLM setup for local evaluation (faster, may be less accurate)
187
+ """
188
+
189
+ name = "eval_delegate"
190
+ category = "eval_delegate"
191
+ description = "Delegates to evaluation templates"
192
+ default_action = ScannerAction.BLOCK
193
+
194
+ def __init__(
195
+ self,
196
+ categories: Optional[List[EvalCategory]] = None,
197
+ thresholds: Optional[Dict[EvalCategory, float]] = None,
198
+ prefer_local: bool = True,
199
+ api_key: Optional[str] = None,
200
+ timeout: int = 30,
201
+ parallel: bool = True,
202
+ aggregation: str = "any",
203
+ action: Optional[ScannerAction] = None,
204
+ enabled: bool = True,
205
+ ):
206
+ """
207
+ Initialize the eval delegate scanner.
208
+
209
+ Args:
210
+ categories: List of eval categories to check
211
+ thresholds: Custom thresholds per category
212
+ prefer_local: Use local evaluator if available
213
+ api_key: API key for cloud evaluation
214
+ timeout: Timeout in seconds
215
+ parallel: Run categories in parallel
216
+ aggregation: "any" or "all" for failure aggregation
217
+ action: Action to take on detection
218
+ enabled: Whether scanner is enabled
219
+ """
220
+ super().__init__(action=action, enabled=enabled)
221
+
222
+ self.categories = categories or [EvalCategory.TOXICITY]
223
+ self.thresholds = thresholds or {}
224
+ self.prefer_local = prefer_local
225
+ self.api_key = api_key
226
+ self.timeout = timeout
227
+ self.parallel = parallel
228
+ self.aggregation = aggregation
229
+
230
+ # Lazy-loaded evaluators
231
+ self._local_evaluator = None
232
+ self._cloud_evaluator = None
233
+
234
+ @classmethod
235
+ def for_toxicity(cls, threshold: float = 0.5, **kwargs) -> "EvalDelegateScanner":
236
+ """Create a scanner for toxicity detection."""
237
+ return cls(
238
+ categories=[EvalCategory.TOXICITY],
239
+ thresholds={EvalCategory.TOXICITY: threshold},
240
+ **kwargs
241
+ )
242
+
243
+ @classmethod
244
+ def for_pii(cls, threshold: float = 0.5, **kwargs) -> "EvalDelegateScanner":
245
+ """Create a scanner for PII detection."""
246
+ return cls(
247
+ categories=[EvalCategory.PII],
248
+ thresholds={EvalCategory.PII: threshold},
249
+ **kwargs
250
+ )
251
+
252
+ @classmethod
253
+ def for_prompt_injection(cls, threshold: float = 0.5, **kwargs) -> "EvalDelegateScanner":
254
+ """Create a scanner for prompt injection detection."""
255
+ return cls(
256
+ categories=[EvalCategory.PROMPT_INJECTION],
257
+ thresholds={EvalCategory.PROMPT_INJECTION: threshold},
258
+ **kwargs
259
+ )
260
+
261
+ @classmethod
262
+ def for_bias(cls, include_specific: bool = True, threshold: float = 0.5, **kwargs) -> "EvalDelegateScanner":
263
+ """
264
+ Create a scanner for bias detection.
265
+
266
+ Args:
267
+ include_specific: Include racial, gender, age bias checks
268
+ threshold: Detection threshold
269
+ """
270
+ categories = [EvalCategory.BIAS]
271
+ if include_specific:
272
+ categories.extend([
273
+ EvalCategory.RACIAL_BIAS,
274
+ EvalCategory.GENDER_BIAS,
275
+ EvalCategory.AGE_BIAS,
276
+ ])
277
+
278
+ thresholds = {cat: threshold for cat in categories}
279
+ return cls(categories=categories, thresholds=thresholds, **kwargs)
280
+
281
+ @classmethod
282
+ def for_safety(cls, threshold: float = 0.5, **kwargs) -> "EvalDelegateScanner":
283
+ """
284
+ Create a comprehensive safety scanner.
285
+
286
+ Includes: toxicity, PII, prompt injection, content safety, NSFW
287
+ """
288
+ categories = [
289
+ EvalCategory.TOXICITY,
290
+ EvalCategory.PII,
291
+ EvalCategory.PROMPT_INJECTION,
292
+ EvalCategory.CONTENT_SAFETY,
293
+ EvalCategory.NSFW,
294
+ ]
295
+ thresholds = {cat: threshold for cat in categories}
296
+ return cls(categories=categories, thresholds=thresholds, **kwargs)
297
+
298
+ @classmethod
299
+ def for_content_moderation(cls, threshold: float = 0.5, **kwargs) -> "EvalDelegateScanner":
300
+ """
301
+ Create a content moderation scanner.
302
+
303
+ Includes: toxicity, NSFW, sexist, content safety
304
+ """
305
+ categories = [
306
+ EvalCategory.TOXICITY,
307
+ EvalCategory.NSFW,
308
+ EvalCategory.SEXIST,
309
+ EvalCategory.CONTENT_SAFETY,
310
+ ]
311
+ thresholds = {cat: threshold for cat in categories}
312
+ return cls(categories=categories, thresholds=thresholds, **kwargs)
313
+
314
+ def _get_local_evaluator(self):
315
+ """Get or create local evaluator."""
316
+ if self._local_evaluator is None:
317
+ try:
318
+ from fi.evals.local import LocalEvaluator
319
+ self._local_evaluator = LocalEvaluator()
320
+ except ImportError:
321
+ self._local_evaluator = None
322
+ return self._local_evaluator
323
+
324
+ def _get_cloud_evaluator(self):
325
+ """Get or create cloud evaluator."""
326
+ if self._cloud_evaluator is None:
327
+ try:
328
+ from fi.evals.evaluator import Evaluator
329
+ self._cloud_evaluator = Evaluator(fi_api_key=self.api_key)
330
+ except Exception:
331
+ self._cloud_evaluator = None
332
+ return self._cloud_evaluator
333
+
334
+ def _get_threshold(self, category: EvalCategory) -> float:
335
+ """Get threshold for a category."""
336
+ if category in self.thresholds:
337
+ return self.thresholds[category]
338
+ return EVAL_TEMPLATE_MAP[category].get("threshold", 0.5)
339
+
340
+ def _evaluate_category(
341
+ self,
342
+ content: str,
343
+ category: EvalCategory,
344
+ context: Optional[str] = None,
345
+ ) -> Dict[str, Any]:
346
+ """
347
+ Evaluate content for a single category.
348
+
349
+ Returns:
350
+ Dict with keys: passed, score, reason, latency_ms
351
+ """
352
+ template_info = EVAL_TEMPLATE_MAP[category]
353
+ eval_name = template_info["eval_name"]
354
+ threshold = self._get_threshold(category)
355
+ invert = template_info.get("invert", True)
356
+
357
+ start_time = time.perf_counter()
358
+
359
+ # Prepare input
360
+ eval_input = {"response": content}
361
+ if context:
362
+ eval_input["context"] = context
363
+
364
+ # Try local evaluation first if preferred
365
+ if self.prefer_local:
366
+ local_eval = self._get_local_evaluator()
367
+ if local_eval and local_eval.can_run_locally(eval_name):
368
+ try:
369
+ result = local_eval.evaluate(
370
+ metric_name=eval_name,
371
+ inputs=[eval_input],
372
+ )
373
+ if result.results.eval_results:
374
+ eval_result = result.results.eval_results[0]
375
+ score = eval_result.output or 0.0
376
+ latency = (time.perf_counter() - start_time) * 1000
377
+
378
+ # Determine pass/fail
379
+ if invert:
380
+ passed = score < threshold
381
+ else:
382
+ passed = score >= threshold
383
+
384
+ return {
385
+ "passed": passed,
386
+ "score": score,
387
+ "reason": eval_result.reason or f"{eval_name}: {score:.2f}",
388
+ "latency_ms": latency,
389
+ "source": "local",
390
+ }
391
+ except Exception:
392
+ # Fall through to cloud evaluation
393
+ pass
394
+
395
+ # Try cloud evaluation
396
+ cloud_eval = self._get_cloud_evaluator()
397
+ if cloud_eval:
398
+ try:
399
+ result = cloud_eval.evaluate(
400
+ eval_templates=[eval_name],
401
+ inputs=[eval_input],
402
+ )
403
+ if result.eval_results:
404
+ eval_result = result.eval_results[0]
405
+ score = eval_result.output or 0.0
406
+ latency = (time.perf_counter() - start_time) * 1000
407
+
408
+ if invert:
409
+ passed = score < threshold
410
+ else:
411
+ passed = score >= threshold
412
+
413
+ return {
414
+ "passed": passed,
415
+ "score": score,
416
+ "reason": eval_result.reason or f"{eval_name}: {score:.2f}",
417
+ "latency_ms": latency,
418
+ "source": "cloud",
419
+ }
420
+ except Exception as e:
421
+ latency = (time.perf_counter() - start_time) * 1000
422
+ return {
423
+ "passed": True, # Default to pass on error
424
+ "score": 0.0,
425
+ "reason": f"Evaluation error: {str(e)}",
426
+ "latency_ms": latency,
427
+ "source": "error",
428
+ }
429
+
430
+ # No evaluator available
431
+ latency = (time.perf_counter() - start_time) * 1000
432
+ return {
433
+ "passed": True,
434
+ "score": 0.0,
435
+ "reason": "No evaluator available",
436
+ "latency_ms": latency,
437
+ "source": "none",
438
+ }
439
+
440
+ def scan(self, content: str, context: Optional[str] = None) -> ScanResult:
441
+ """
442
+ Scan content using delegated evaluations.
443
+
444
+ Args:
445
+ content: Content to scan
446
+ context: Optional context
447
+
448
+ Returns:
449
+ ScanResult with aggregated results from all categories
450
+ """
451
+ if not self.enabled:
452
+ return self._create_result(
453
+ passed=True,
454
+ reason="Scanner disabled",
455
+ )
456
+
457
+ start_time = time.perf_counter()
458
+ matches = []
459
+ category_results = {}
460
+
461
+ # Evaluate each category
462
+ if self.parallel and len(self.categories) > 1:
463
+ # Parallel execution
464
+ from concurrent.futures import ThreadPoolExecutor, as_completed
465
+
466
+ with ThreadPoolExecutor(max_workers=len(self.categories)) as executor:
467
+ futures = {
468
+ executor.submit(self._evaluate_category, content, cat, context): cat
469
+ for cat in self.categories
470
+ }
471
+
472
+ for future in as_completed(futures, timeout=self.timeout):
473
+ cat = futures[future]
474
+ try:
475
+ category_results[cat] = future.result()
476
+ except Exception as e:
477
+ category_results[cat] = {
478
+ "passed": True,
479
+ "score": 0.0,
480
+ "reason": f"Timeout/error: {str(e)}",
481
+ "latency_ms": 0,
482
+ "source": "error",
483
+ }
484
+ else:
485
+ # Sequential execution
486
+ for cat in self.categories:
487
+ category_results[cat] = self._evaluate_category(content, cat, context)
488
+
489
+ # Aggregate results
490
+ failed_categories = []
491
+ total_score = 0.0
492
+
493
+ for cat, result in category_results.items():
494
+ total_score += result["score"]
495
+
496
+ if not result["passed"]:
497
+ failed_categories.append(cat)
498
+ template_info = EVAL_TEMPLATE_MAP[cat]
499
+ matches.append(ScanMatch(
500
+ pattern_name=f"eval:{cat.value}",
501
+ matched_text=content[:100] + "..." if len(content) > 100 else content,
502
+ start=0,
503
+ end=len(content),
504
+ confidence=result["score"],
505
+ metadata={
506
+ "category": cat.value,
507
+ "eval_name": template_info["eval_name"],
508
+ "threshold": self._get_threshold(cat),
509
+ "source": result.get("source", "unknown"),
510
+ },
511
+ ))
512
+
513
+ # Determine overall pass/fail
514
+ if self.aggregation == "any":
515
+ passed = len(failed_categories) == 0
516
+ else: # "all"
517
+ passed = len(failed_categories) < len(self.categories)
518
+
519
+ # Calculate average score
520
+ avg_score = total_score / len(self.categories) if self.categories else 0.0
521
+
522
+ # Generate reason
523
+ if passed:
524
+ reason = f"All {len(self.categories)} eval checks passed"
525
+ else:
526
+ failed_names = [c.value for c in failed_categories]
527
+ reason = f"Failed eval checks: {', '.join(failed_names)}"
528
+
529
+ total_latency = (time.perf_counter() - start_time) * 1000
530
+
531
+ return self._create_result(
532
+ passed=passed,
533
+ matches=matches,
534
+ score=avg_score,
535
+ reason=reason,
536
+ latency_ms=total_latency,
537
+ metadata={
538
+ "categories_checked": [c.value for c in self.categories],
539
+ "categories_failed": [c.value for c in failed_categories],
540
+ "category_results": {
541
+ c.value: {
542
+ "passed": r["passed"],
543
+ "score": r["score"],
544
+ "source": r.get("source", "unknown"),
545
+ }
546
+ for c, r in category_results.items()
547
+ },
548
+ },
549
+ )
550
+
551
+
552
+ # Convenience aliases
553
+ def PIIScanner(**kwargs):
554
+ return EvalDelegateScanner.for_pii(**kwargs)
555
+
556
+
557
+ def ToxicityScanner(**kwargs):
558
+ return EvalDelegateScanner.for_toxicity(**kwargs)
559
+
560
+
561
+ def PromptInjectionScanner(**kwargs):
562
+ return EvalDelegateScanner.for_prompt_injection(**kwargs)
563
+
564
+
565
+ def BiasScanner(**kwargs):
566
+ return EvalDelegateScanner.for_bias(**kwargs)
567
+
568
+
569
+ def SafetyScanner(**kwargs):
570
+ return EvalDelegateScanner.for_safety(**kwargs)
571
+
572
+
573
+ def ContentModerationScanner(**kwargs):
574
+ return EvalDelegateScanner.for_content_moderation(**kwargs)