agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,625 @@
1
+ """AutoEval Pipeline - Main API for automatic evaluation pipelines.
2
+
3
+ Provides the core AutoEvalPipeline class for creating and running
4
+ evaluation pipelines from natural language descriptions or templates.
5
+ """
6
+
7
+ import logging
8
+ import time
9
+ from typing import Dict, Any, List, Optional, Type, Union
10
+
11
+ from .types import AutoEvalResult, AppAnalysis
12
+ from .config import AutoEvalConfig, EvalConfig, ScannerConfig
13
+ from .analyzer import AppAnalyzer
14
+ from .recommender import EvalRecommender
15
+ from .templates import get_template, list_templates
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+
20
+ # Registry for eval/scanner class lookups
21
+ _EVAL_CLASS_REGISTRY: Dict[str, Type] = {}
22
+ _SCANNER_CLASS_REGISTRY: Dict[str, Type] = {}
23
+
24
+
25
+ def register_eval_class(name: str, cls: Type) -> None:
26
+ """Register an evaluation class for AutoEval lookup."""
27
+ _EVAL_CLASS_REGISTRY[name] = cls
28
+
29
+
30
+ def register_scanner_class(name: str, cls: Type) -> None:
31
+ """Register a scanner class for AutoEval lookup."""
32
+ _SCANNER_CLASS_REGISTRY[name] = cls
33
+
34
+
35
+ def _get_eval_class(name: str) -> Optional[Type]:
36
+ """Get evaluation class by name."""
37
+ # Try direct registry lookup first
38
+ if name in _EVAL_CLASS_REGISTRY:
39
+ return _EVAL_CLASS_REGISTRY[name]
40
+
41
+ # Try to import from framework
42
+ try:
43
+ from fi.evals.framework import EvalRegistry
44
+
45
+ if EvalRegistry.is_registered(name):
46
+ return EvalRegistry.get(name)
47
+ except (ImportError, ValueError):
48
+ pass
49
+
50
+ # Try to import from evals module
51
+ try:
52
+ from fi.evals import templates as eval_templates
53
+
54
+ if hasattr(eval_templates, name):
55
+ return getattr(eval_templates, name)
56
+ except ImportError:
57
+ pass
58
+
59
+ return None
60
+
61
+
62
+ def _get_scanner_class(name: str) -> Optional[Type]:
63
+ """Get scanner class by name."""
64
+ # Try direct registry lookup first
65
+ if name in _SCANNER_CLASS_REGISTRY:
66
+ return _SCANNER_CLASS_REGISTRY[name]
67
+
68
+ # Try to import from guardrails.scanners
69
+ try:
70
+ from fi.evals.guardrails import scanners as scanner_module
71
+
72
+ if hasattr(scanner_module, name):
73
+ return getattr(scanner_module, name)
74
+ except ImportError:
75
+ pass
76
+
77
+ return None
78
+
79
+
80
+ class AutoEvalPipeline:
81
+ """
82
+ Automatic evaluation pipeline builder.
83
+
84
+ Creates evaluation pipelines from natural language descriptions,
85
+ pre-built templates, or manual configuration.
86
+
87
+ Example:
88
+ # From natural language description
89
+ pipeline = AutoEvalPipeline.from_description(
90
+ "A RAG-based customer support chatbot for healthcare. "
91
+ "Retrieves patient records and answers questions about appointments."
92
+ )
93
+
94
+ # From pre-built template
95
+ pipeline = AutoEvalPipeline.from_template("rag_system")
96
+
97
+ # Run evaluation
98
+ result = pipeline.evaluate({
99
+ "query": "When is my appointment?",
100
+ "response": "Your appointment is Monday at 2pm.",
101
+ "context": "Patient has appointment on 2024-01-15 14:00",
102
+ })
103
+
104
+ print(result.passed) # True/False
105
+ print(result.explain()) # Detailed breakdown
106
+
107
+ # Export configuration
108
+ pipeline.export_yaml("eval_config.yaml")
109
+ """
110
+
111
+ def __init__(
112
+ self,
113
+ config: AutoEvalConfig,
114
+ analysis: Optional[AppAnalysis] = None,
115
+ ):
116
+ """
117
+ Initialize the pipeline with configuration.
118
+
119
+ Args:
120
+ config: AutoEvalConfig with evaluations and scanners
121
+ analysis: Optional AppAnalysis for explanation context
122
+ """
123
+ self.config = config
124
+ self.analysis = analysis
125
+ self._evaluator = None
126
+ self._scanner_pipeline = None
127
+ self._eval_instances: List[Any] = []
128
+ self._scanner_instances: List[Any] = []
129
+
130
+ @classmethod
131
+ def from_description(
132
+ cls,
133
+ description: str,
134
+ llm_provider: Optional[Any] = None,
135
+ name: Optional[str] = None,
136
+ ) -> "AutoEvalPipeline":
137
+ """
138
+ Create pipeline from natural language description.
139
+
140
+ Uses LLM-powered analysis when available, falls back to
141
+ rule-based analysis otherwise.
142
+
143
+ Args:
144
+ description: Natural language application description
145
+ llm_provider: Optional LLM provider for intelligent analysis
146
+ name: Optional name for the pipeline
147
+
148
+ Returns:
149
+ Configured AutoEvalPipeline
150
+
151
+ Example:
152
+ pipeline = AutoEvalPipeline.from_description(
153
+ "A customer support chatbot for a healthcare company. "
154
+ "It retrieves patient information and answers questions."
155
+ )
156
+ """
157
+ # Analyze the description
158
+ analyzer = AppAnalyzer(llm_provider=llm_provider)
159
+ analysis = analyzer.analyze(description)
160
+
161
+ # Generate recommendations
162
+ recommender = EvalRecommender()
163
+ evals, scanners = recommender.recommend(analysis)
164
+
165
+ # Build config
166
+ config = AutoEvalConfig(
167
+ name=name or f"autoeval_{analysis.category.value}",
168
+ description=description[:200] if len(description) > 200 else description,
169
+ app_category=analysis.category.value,
170
+ risk_level=analysis.risk_level.value,
171
+ domain_sensitivity=analysis.domain_sensitivity.value,
172
+ evaluations=evals,
173
+ scanners=scanners,
174
+ )
175
+
176
+ return cls(config, analysis)
177
+
178
+ @classmethod
179
+ def from_template(cls, template_name: str) -> "AutoEvalPipeline":
180
+ """
181
+ Create pipeline from pre-built template.
182
+
183
+ Available templates:
184
+ - customer_support: Customer service chatbots
185
+ - rag_system: RAG-based document Q&A
186
+ - code_assistant: Code generation and review
187
+ - content_moderation: Content filtering and safety
188
+ - agent_workflow: Autonomous agents with tool use
189
+ - healthcare: Healthcare applications (HIPAA)
190
+ - financial: Financial services
191
+
192
+ Args:
193
+ template_name: Name of the template to use
194
+
195
+ Returns:
196
+ Configured AutoEvalPipeline
197
+
198
+ Raises:
199
+ ValueError: If template not found
200
+
201
+ Example:
202
+ pipeline = AutoEvalPipeline.from_template("rag_system")
203
+ """
204
+ config = get_template(template_name)
205
+ if config is None:
206
+ available = list(list_templates().keys())
207
+ raise ValueError(
208
+ f"Template '{template_name}' not found. "
209
+ f"Available templates: {available}"
210
+ )
211
+ return cls(config)
212
+
213
+ @classmethod
214
+ def from_config(cls, config: AutoEvalConfig) -> "AutoEvalPipeline":
215
+ """
216
+ Create pipeline from existing configuration.
217
+
218
+ Args:
219
+ config: AutoEvalConfig instance
220
+
221
+ Returns:
222
+ Configured AutoEvalPipeline
223
+ """
224
+ return cls(config)
225
+
226
+ @classmethod
227
+ def from_yaml(cls, path: str) -> "AutoEvalPipeline":
228
+ """
229
+ Load pipeline from YAML file.
230
+
231
+ Args:
232
+ path: Path to YAML configuration file
233
+
234
+ Returns:
235
+ Configured AutoEvalPipeline
236
+ """
237
+ from .export import load_config
238
+
239
+ config = load_config(path)
240
+ return cls(config)
241
+
242
+ def _is_core_metric(self, name: str) -> bool:
243
+ """Check if a name corresponds to a core metric in the local registry."""
244
+ try:
245
+ from fi.evals.local.registry import get_registry
246
+ return get_registry().is_registered(name)
247
+ except ImportError:
248
+ return False
249
+
250
+ def _get_metric_configs(self) -> List[EvalConfig]:
251
+ """Get eval configs that are core metrics (routed through evaluate())."""
252
+ return [
253
+ e for e in self.config.evaluations
254
+ if e.enabled and self._is_core_metric(e.name)
255
+ ]
256
+
257
+ def _get_class_configs(self) -> List[EvalConfig]:
258
+ """Get eval configs that are framework classes (routed through Evaluator)."""
259
+ return [
260
+ e for e in self.config.evaluations
261
+ if e.enabled and not self._is_core_metric(e.name)
262
+ ]
263
+
264
+ def _run_core_metrics(
265
+ self,
266
+ inputs: Dict[str, Any],
267
+ feedback_store: Optional[Any] = None,
268
+ ) -> List[Any]:
269
+ """Run core metrics via evaluate() API and return EvalResults."""
270
+ from fi.evals import evaluate as core_evaluate
271
+
272
+ metric_configs = self._get_metric_configs()
273
+ if not metric_configs:
274
+ return []
275
+
276
+ results = []
277
+ for eval_config in metric_configs:
278
+ try:
279
+ result = core_evaluate(
280
+ eval_config.name,
281
+ model=eval_config.model,
282
+ augment=eval_config.augment,
283
+ feedback_store=feedback_store,
284
+ **inputs,
285
+ )
286
+ results.append(result)
287
+ except Exception as e:
288
+ logger.warning(f"Core metric '{eval_config.name}' failed: {e}")
289
+ return results
290
+
291
+ def _build_evaluator(self) -> None:
292
+ """Build the evaluator with framework-class evaluations only."""
293
+ if self._evaluator is not None:
294
+ return
295
+
296
+ from fi.evals.framework import Evaluator, ExecutionMode
297
+
298
+ # Only build for non-metric (framework class) evals
299
+ self._eval_instances = []
300
+ for eval_config in self._get_class_configs():
301
+ eval_class = _get_eval_class(eval_config.name)
302
+ if eval_class is None:
303
+ logger.warning(f"Evaluation class not found: {eval_config.name}")
304
+ continue
305
+
306
+ try:
307
+ instance = eval_class(**eval_config.params)
308
+ self._eval_instances.append(instance)
309
+ except Exception as e:
310
+ logger.warning(f"Failed to instantiate {eval_config.name}: {e}")
311
+
312
+ if self._eval_instances:
313
+ mode = (
314
+ ExecutionMode.NON_BLOCKING
315
+ if self.config.execution_mode == "non_blocking"
316
+ else ExecutionMode.BLOCKING
317
+ )
318
+ self._evaluator = Evaluator(
319
+ evaluations=self._eval_instances,
320
+ mode=mode,
321
+ max_workers=self.config.parallel_workers,
322
+ fail_fast=self.config.fail_fast,
323
+ )
324
+
325
+ def _build_scanner_pipeline(self) -> None:
326
+ """Build the scanner pipeline with configured scanners."""
327
+ if self._scanner_pipeline is not None:
328
+ return
329
+
330
+ from fi.evals.guardrails.scanners import ScannerPipeline
331
+
332
+ # Instantiate scanner classes
333
+ self._scanner_instances = []
334
+ for scanner_config in self.config.scanners:
335
+ if not scanner_config.enabled:
336
+ continue
337
+
338
+ scanner_class = _get_scanner_class(scanner_config.name)
339
+ if scanner_class is None:
340
+ logger.warning(f"Scanner class not found: {scanner_config.name}")
341
+ continue
342
+
343
+ try:
344
+ instance = scanner_class(**scanner_config.params)
345
+ # Set threshold and action if the scanner supports them
346
+ if hasattr(instance, "threshold"):
347
+ instance.threshold = scanner_config.threshold
348
+ if hasattr(instance, "action"):
349
+ from fi.evals.guardrails.scanners.base import ScannerAction
350
+
351
+ instance.action = ScannerAction(scanner_config.action)
352
+ self._scanner_instances.append(instance)
353
+ except Exception as e:
354
+ logger.warning(f"Failed to instantiate {scanner_config.name}: {e}")
355
+
356
+ if self._scanner_instances:
357
+ self._scanner_pipeline = ScannerPipeline(
358
+ scanners=self._scanner_instances,
359
+ parallel=True,
360
+ fail_fast=self.config.fail_fast,
361
+ )
362
+
363
+ def evaluate(
364
+ self,
365
+ inputs: Dict[str, Any],
366
+ scan_content: Optional[str] = None,
367
+ feedback_store: Optional[Any] = None,
368
+ ) -> AutoEvalResult:
369
+ """
370
+ Run the full evaluation pipeline.
371
+
372
+ Executes scanners first (fast), then evaluations.
373
+ If scanners block, evaluations may be skipped.
374
+
375
+ Args:
376
+ inputs: Input data for evaluations (e.g., query, response, context)
377
+ scan_content: Content to scan (defaults to response from inputs)
378
+
379
+ Returns:
380
+ AutoEvalResult with combined results
381
+
382
+ Example:
383
+ result = pipeline.evaluate({
384
+ "query": "What is the patient's blood type?",
385
+ "response": "The patient's blood type is O+.",
386
+ "context": "Medical record: Blood type O+",
387
+ })
388
+
389
+ if result.passed:
390
+ print("Evaluation passed!")
391
+ else:
392
+ print(f"Failed: {result.explain()}")
393
+ """
394
+ start_time = time.perf_counter()
395
+
396
+ # Build components lazily
397
+ self._build_evaluator()
398
+ self._build_scanner_pipeline()
399
+
400
+ scan_result = None
401
+ eval_result = None
402
+ metric_results = []
403
+ blocked_by_scanner = False
404
+
405
+ # Run scanners first (fast)
406
+ if self._scanner_pipeline:
407
+ content = scan_content or inputs.get("response", "")
408
+ context = inputs.get("context")
409
+ scan_result = self._scanner_pipeline.scan(content, context)
410
+ blocked_by_scanner = not scan_result.passed
411
+
412
+ # Run evaluations if not blocked
413
+ if not blocked_by_scanner:
414
+ # Core metrics via evaluate() API
415
+ metric_results = self._run_core_metrics(inputs, feedback_store=feedback_store)
416
+
417
+ # Framework class evals via Evaluator
418
+ if self._evaluator:
419
+ eval_result = self._evaluator.run(inputs)
420
+
421
+ # Determine overall pass/fail
422
+ passed = True
423
+ if scan_result and not scan_result.passed:
424
+ passed = False
425
+
426
+ # Check core metric results against thresholds
427
+ if metric_results:
428
+ metric_configs = {e.name: e for e in self._get_metric_configs()}
429
+ failed_count = 0
430
+ total_count = len(metric_results)
431
+ for r in metric_results:
432
+ config = metric_configs.get(getattr(r, "eval_name", ""))
433
+ threshold = config.threshold if config else 0.5
434
+ score = getattr(r, "score", None)
435
+ if score is not None and score < threshold:
436
+ failed_count += 1
437
+ if total_count > 0:
438
+ success_rate = (total_count - failed_count) / total_count
439
+ if success_rate < self.config.global_pass_rate:
440
+ passed = False
441
+
442
+ if eval_result:
443
+ # Check if framework evaluations meet threshold
444
+ batch = eval_result.wait() if eval_result.is_future else eval_result.batch
445
+ if batch and batch.success_rate < self.config.global_pass_rate:
446
+ passed = False
447
+
448
+ total_latency = (time.perf_counter() - start_time) * 1000
449
+
450
+ return AutoEvalResult(
451
+ passed=passed,
452
+ scan_result=scan_result,
453
+ eval_result=eval_result,
454
+ metric_results=metric_results,
455
+ blocked_by_scanner=blocked_by_scanner,
456
+ total_latency_ms=total_latency,
457
+ )
458
+
459
+ def add(self, item: Union[EvalConfig, ScannerConfig]) -> "AutoEvalPipeline":
460
+ """
461
+ Add an evaluation or scanner to the pipeline.
462
+
463
+ Args:
464
+ item: EvalConfig or ScannerConfig to add
465
+
466
+ Returns:
467
+ Self for chaining
468
+
469
+ Example:
470
+ pipeline.add(EvalConfig("CustomEval", threshold=0.8))
471
+ pipeline.add(ScannerConfig("CustomScanner", action="flag"))
472
+ """
473
+ if isinstance(item, EvalConfig):
474
+ self.config.evaluations.append(item)
475
+ self._evaluator = None # Reset to rebuild
476
+ elif isinstance(item, ScannerConfig):
477
+ self.config.scanners.append(item)
478
+ self._scanner_pipeline = None # Reset to rebuild
479
+ return self
480
+
481
+ def remove(self, name: str) -> "AutoEvalPipeline":
482
+ """
483
+ Remove an evaluation or scanner by name.
484
+
485
+ Args:
486
+ name: Name of the evaluation or scanner to remove
487
+
488
+ Returns:
489
+ Self for chaining
490
+
491
+ Example:
492
+ pipeline.remove("CoherenceEval")
493
+ """
494
+ # Try to remove from evaluations
495
+ original_count = len(self.config.evaluations)
496
+ self.config.evaluations = [
497
+ e for e in self.config.evaluations if e.name != name
498
+ ]
499
+ if len(self.config.evaluations) < original_count:
500
+ self._evaluator = None
501
+
502
+ # Try to remove from scanners
503
+ original_count = len(self.config.scanners)
504
+ self.config.scanners = [s for s in self.config.scanners if s.name != name]
505
+ if len(self.config.scanners) < original_count:
506
+ self._scanner_pipeline = None
507
+
508
+ return self
509
+
510
+ def set_threshold(self, name: str, threshold: float) -> "AutoEvalPipeline":
511
+ """
512
+ Set threshold for an evaluation or scanner.
513
+
514
+ Args:
515
+ name: Name of the evaluation or scanner
516
+ threshold: New threshold value (0.0-1.0)
517
+
518
+ Returns:
519
+ Self for chaining
520
+
521
+ Example:
522
+ pipeline.set_threshold("CoherenceEval", 0.9)
523
+ """
524
+ for e in self.config.evaluations:
525
+ if e.name == name:
526
+ e.threshold = threshold
527
+ self._evaluator = None
528
+
529
+ for s in self.config.scanners:
530
+ if s.name == name:
531
+ s.threshold = threshold
532
+ self._scanner_pipeline = None
533
+
534
+ return self
535
+
536
+ def enable(self, name: str) -> "AutoEvalPipeline":
537
+ """Enable an evaluation or scanner."""
538
+ for e in self.config.evaluations:
539
+ if e.name == name:
540
+ e.enabled = True
541
+ self._evaluator = None
542
+
543
+ for s in self.config.scanners:
544
+ if s.name == name:
545
+ s.enabled = True
546
+ self._scanner_pipeline = None
547
+
548
+ return self
549
+
550
+ def disable(self, name: str) -> "AutoEvalPipeline":
551
+ """Disable an evaluation or scanner."""
552
+ for e in self.config.evaluations:
553
+ if e.name == name:
554
+ e.enabled = False
555
+ self._evaluator = None
556
+
557
+ for s in self.config.scanners:
558
+ if s.name == name:
559
+ s.enabled = False
560
+ self._scanner_pipeline = None
561
+
562
+ return self
563
+
564
+ def export_yaml(self, path: str) -> None:
565
+ """
566
+ Export pipeline configuration to YAML file.
567
+
568
+ Args:
569
+ path: Path to write YAML file
570
+
571
+ Example:
572
+ pipeline.export_yaml("eval_config.yaml")
573
+ """
574
+ from .export import export_yaml
575
+
576
+ export_yaml(self.config, path)
577
+
578
+ def export_json(self, path: str) -> None:
579
+ """
580
+ Export pipeline configuration to JSON file.
581
+
582
+ Args:
583
+ path: Path to write JSON file
584
+ """
585
+ from .export import export_json
586
+
587
+ export_json(self.config, path)
588
+
589
+ def explain(self) -> str:
590
+ """
591
+ Get human-readable explanation of the pipeline.
592
+
593
+ Returns:
594
+ Detailed explanation string
595
+
596
+ Example:
597
+ print(pipeline.explain())
598
+ """
599
+ lines = [self.config.summary()]
600
+
601
+ if self.analysis:
602
+ lines.append("")
603
+ lines.append("Analysis Details:")
604
+ lines.append(f" Confidence: {self.analysis.confidence:.0%}")
605
+ lines.append(f" Explanation: {self.analysis.explanation}")
606
+ if self.analysis.detected_features:
607
+ lines.append(f" Detected Features: {', '.join(self.analysis.detected_features)}")
608
+
609
+ return "\n".join(lines)
610
+
611
+ def summary(self) -> str:
612
+ """Get brief summary of the pipeline."""
613
+ return (
614
+ f"AutoEvalPipeline: {self.config.name}\n"
615
+ f" Evaluations: {len(self.config.evaluations)}\n"
616
+ f" Scanners: {len(self.config.scanners)}\n"
617
+ f" Risk Level: {self.config.risk_level}"
618
+ )
619
+
620
+ def __repr__(self) -> str:
621
+ return (
622
+ f"AutoEvalPipeline(name={self.config.name!r}, "
623
+ f"evaluations={len(self.config.evaluations)}, "
624
+ f"scanners={len(self.config.scanners)})"
625
+ )