agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,647 @@
1
+ """
2
+ Unified Evaluator API.
3
+
4
+ Provides a single interface for running evaluations in any mode:
5
+ - Blocking: Synchronous execution, waits for results
6
+ - Non-blocking: Background execution, zero latency
7
+ - Distributed: Scalable execution via pluggable backends
8
+
9
+ Example:
10
+ from fi.evals.framework import Evaluator, ExecutionMode
11
+
12
+ # Create evaluator
13
+ evaluator = Evaluator(
14
+ evaluations=[ToxicityEval(), BiasEval()],
15
+ mode=ExecutionMode.NON_BLOCKING,
16
+ )
17
+
18
+ # Run evaluations (returns immediately in non-blocking mode)
19
+ result = evaluator.run({"response": "..."})
20
+
21
+ # Get results when needed
22
+ if result.is_future:
23
+ batch = result.wait()
24
+ else:
25
+ batch = result.batch
26
+ """
27
+
28
+ import logging
29
+ import time
30
+ from typing import Dict, Any, List, Optional, Callable
31
+ from dataclasses import dataclass, field
32
+ from datetime import datetime, timezone
33
+
34
+ from .types import ExecutionMode, FrameworkEvalResult, BatchEvalResult, EvalStatus
35
+ from .context import EvalContext
36
+ from .protocols import BaseEvaluation, EvalRegistry
37
+ from .evaluators.blocking import BlockingEvaluator
38
+ from .evaluators.non_blocking import (
39
+ NonBlockingEvaluator,
40
+ BatchEvalFuture,
41
+ )
42
+ from .backends import Backend, ThreadPoolBackend
43
+
44
+ logger = logging.getLogger(__name__)
45
+
46
+ # Internal alias for brevity — this is the framework-level EvalResult
47
+ EvalResult = FrameworkEvalResult
48
+
49
+
50
+ @dataclass
51
+ class EvaluatorResult:
52
+ """
53
+ Result from an evaluator run.
54
+
55
+ Wraps either immediate results (blocking) or futures (non-blocking).
56
+
57
+ Attributes:
58
+ batch: Immediate BatchEvalResult (blocking mode)
59
+ future: BatchEvalFuture for async results (non-blocking mode)
60
+ mode: The execution mode used
61
+ submitted_at: When the evaluation was submitted
62
+ """
63
+
64
+ batch: Optional[BatchEvalResult] = None
65
+ future: Optional[BatchEvalFuture] = None
66
+ mode: ExecutionMode = ExecutionMode.BLOCKING
67
+ submitted_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
68
+
69
+ @property
70
+ def is_future(self) -> bool:
71
+ """Whether this result is a future (non-blocking)."""
72
+ return self.future is not None
73
+
74
+ @property
75
+ def is_ready(self) -> bool:
76
+ """Whether results are ready."""
77
+ if self.batch is not None:
78
+ return True
79
+ if self.future is not None:
80
+ return self.future.done()
81
+ return False
82
+
83
+ def wait(self, timeout: Optional[float] = None) -> BatchEvalResult:
84
+ """
85
+ Wait for and return results.
86
+
87
+ Args:
88
+ timeout: Maximum seconds to wait (non-blocking only)
89
+
90
+ Returns:
91
+ BatchEvalResult with all evaluation results
92
+ """
93
+ if self.batch is not None:
94
+ return self.batch
95
+ if self.future is not None:
96
+ return self.future.results(timeout=timeout)
97
+ raise ValueError("No results available")
98
+
99
+ @property
100
+ def results(self) -> List[EvalResult]:
101
+ """Get individual results (waits if necessary)."""
102
+ return self.wait().results
103
+
104
+ @property
105
+ def success_rate(self) -> float:
106
+ """Get success rate (waits if necessary)."""
107
+ return self.wait().success_rate
108
+
109
+
110
+ class FrameworkEvaluator:
111
+ """
112
+ Unified evaluator for all execution modes.
113
+
114
+ Provides a consistent interface regardless of how evaluations are executed.
115
+ Supports blocking, non-blocking, and distributed modes with automatic
116
+ span enrichment.
117
+
118
+ Example:
119
+ # Simple blocking usage
120
+ evaluator = FrameworkEvaluator([ToxicityEval()])
121
+ result = evaluator.run({"response": "..."})
122
+ print(f"Score: {result.results[0].value}")
123
+
124
+ # Non-blocking for production
125
+ evaluator = FrameworkEvaluator(
126
+ [ToxicityEval(), BiasEval()],
127
+ mode=ExecutionMode.NON_BLOCKING,
128
+ )
129
+ result = evaluator.run({"response": "..."}) # Returns immediately
130
+ # ... do other work ...
131
+ batch = result.wait() # Get results when needed
132
+
133
+ # With custom backend
134
+ evaluator = FrameworkEvaluator(
135
+ [ToxicityEval()],
136
+ mode=ExecutionMode.NON_BLOCKING,
137
+ backend=MyTemporalBackend(),
138
+ )
139
+
140
+ Thread Safety:
141
+ This class is thread-safe. Multiple threads can call run() concurrently.
142
+ """
143
+
144
+ def __init__(
145
+ self,
146
+ evaluations: Optional[List[BaseEvaluation]] = None,
147
+ mode: ExecutionMode = ExecutionMode.BLOCKING,
148
+ auto_enrich_span: bool = True,
149
+ fail_fast: bool = False,
150
+ validate_inputs: bool = True,
151
+ max_workers: int = 4,
152
+ backend: Optional[Backend] = None,
153
+ ):
154
+ """
155
+ Initialize the evaluator.
156
+
157
+ Args:
158
+ evaluations: List of evaluations to run
159
+ mode: Execution mode (BLOCKING, NON_BLOCKING, DISTRIBUTED)
160
+ auto_enrich_span: Whether to automatically enrich OTEL spans
161
+ fail_fast: Stop on first failure
162
+ validate_inputs: Whether to validate inputs before evaluation
163
+ max_workers: Max concurrent workers (non-blocking/distributed)
164
+ backend: Custom backend for execution (uses ThreadPool if None)
165
+ """
166
+ self.evaluations = list(evaluations) if evaluations else []
167
+ self.mode = mode
168
+ self.auto_enrich_span = auto_enrich_span
169
+ self.fail_fast = fail_fast
170
+ self.validate_inputs = validate_inputs
171
+ self.max_workers = max_workers
172
+ self._backend = backend
173
+
174
+ # Internal evaluators (created lazily)
175
+ self._blocking: Optional[BlockingEvaluator] = None
176
+ self._non_blocking: Optional[NonBlockingEvaluator] = None
177
+
178
+ def add(self, evaluation: BaseEvaluation) -> "FrameworkEvaluator":
179
+ """
180
+ Add an evaluation to run.
181
+
182
+ Args:
183
+ evaluation: The evaluation to add
184
+
185
+ Returns:
186
+ Self for chaining
187
+ """
188
+ self.evaluations.append(evaluation)
189
+ return self
190
+
191
+ def add_by_name(self, name: str, version: str = "latest") -> "FrameworkEvaluator":
192
+ """
193
+ Add an evaluation by name from the registry.
194
+
195
+ Args:
196
+ name: Evaluation name
197
+ version: Version (default: latest)
198
+
199
+ Returns:
200
+ Self for chaining
201
+
202
+ Raises:
203
+ KeyError: If evaluation not found
204
+ """
205
+ eval_class = EvalRegistry.get(name, version)
206
+ if eval_class is None:
207
+ raise KeyError(f"Evaluation not found: {name}@{version}")
208
+ self.evaluations.append(eval_class())
209
+ return self
210
+
211
+ def run(
212
+ self,
213
+ inputs: Dict[str, Any],
214
+ context: Optional[EvalContext] = None,
215
+ callback: Optional[Callable[[EvalResult], None]] = None,
216
+ ) -> EvaluatorResult:
217
+ """
218
+ Run all evaluations on the given inputs.
219
+
220
+ Args:
221
+ inputs: Input data for evaluations
222
+ context: Optional trace context for span enrichment
223
+ callback: Optional callback for each result (non-blocking only)
224
+
225
+ Returns:
226
+ EvaluatorResult wrapping results or future
227
+
228
+ Raises:
229
+ ValueError: If no evaluations configured
230
+ """
231
+ if not self.evaluations:
232
+ raise ValueError("No evaluations configured")
233
+
234
+ if self.mode == ExecutionMode.BLOCKING:
235
+ return self._run_blocking(inputs, context)
236
+ elif self.mode == ExecutionMode.NON_BLOCKING:
237
+ return self._run_non_blocking(inputs, context, callback)
238
+ elif self.mode == ExecutionMode.DISTRIBUTED:
239
+ return self._run_distributed(inputs, context, callback)
240
+ else:
241
+ raise ValueError(f"Unknown execution mode: {self.mode}")
242
+
243
+ def run_single(
244
+ self,
245
+ evaluation: BaseEvaluation,
246
+ inputs: Dict[str, Any],
247
+ context: Optional[EvalContext] = None,
248
+ ) -> EvalResult:
249
+ """
250
+ Run a single evaluation.
251
+
252
+ Always runs in blocking mode for simplicity.
253
+
254
+ Args:
255
+ evaluation: The evaluation to run
256
+ inputs: Input data
257
+ context: Optional trace context
258
+
259
+ Returns:
260
+ Single EvalResult
261
+ """
262
+ evaluator = self._get_blocking_evaluator()
263
+ results = evaluator.evaluate(inputs, evaluations=[evaluation], context=context)
264
+ return results[0]
265
+
266
+ def _run_blocking(
267
+ self,
268
+ inputs: Dict[str, Any],
269
+ context: Optional[EvalContext],
270
+ ) -> EvaluatorResult:
271
+ """Run evaluations in blocking mode."""
272
+ evaluator = self._get_blocking_evaluator()
273
+ results = evaluator.evaluate(inputs, context=context)
274
+ batch = BatchEvalResult.from_results(results)
275
+
276
+ return EvaluatorResult(
277
+ batch=batch,
278
+ mode=ExecutionMode.BLOCKING,
279
+ )
280
+
281
+ def _run_non_blocking(
282
+ self,
283
+ inputs: Dict[str, Any],
284
+ context: Optional[EvalContext],
285
+ callback: Optional[Callable[[EvalResult], None]],
286
+ ) -> EvaluatorResult:
287
+ """Run evaluations in non-blocking mode."""
288
+ evaluator = self._get_non_blocking_evaluator()
289
+ future = evaluator.evaluate(inputs, context=context, callback=callback)
290
+
291
+ return EvaluatorResult(
292
+ future=future,
293
+ mode=ExecutionMode.NON_BLOCKING,
294
+ )
295
+
296
+ def _run_distributed(
297
+ self,
298
+ inputs: Dict[str, Any],
299
+ context: Optional[EvalContext],
300
+ callback: Optional[Callable[[EvalResult], None]],
301
+ ) -> EvaluatorResult:
302
+ """
303
+ Run evaluations in distributed mode.
304
+
305
+ Uses the configured backend for execution. Submits each evaluation
306
+ as a task to the backend, collects results, and returns a BatchEvalResult.
307
+ Falls back to non-blocking if no backend is configured.
308
+ """
309
+ if self._backend is None:
310
+ return self._run_non_blocking(inputs, context, callback)
311
+
312
+ context_dict = None
313
+ if context and hasattr(context, "to_dict"):
314
+ context_dict = context.to_dict()
315
+
316
+ # Submit each evaluation as a task to the backend, collecting handles
317
+ # or immediate failure results
318
+ handles = [] # List of (index, handle) for successful submissions
319
+ results = [None] * len(self.evaluations) # Pre-allocate result slots
320
+
321
+ for i, evaluation in enumerate(self.evaluations):
322
+ eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
323
+ eval_version = getattr(evaluation, "version", "1.0.0")
324
+ try:
325
+ handle = self._backend.submit(
326
+ _execute_single_evaluation,
327
+ args=(evaluation, inputs, self.validate_inputs),
328
+ context=context_dict,
329
+ )
330
+ handles.append((i, handle))
331
+ except Exception as e:
332
+ error_result = EvalResult(
333
+ value=None,
334
+ eval_name=eval_name,
335
+ eval_version=eval_version,
336
+ latency_ms=0.0,
337
+ status=EvalStatus.FAILED,
338
+ error=str(e),
339
+ )
340
+ results[i] = error_result
341
+ if callback:
342
+ callback(error_result)
343
+
344
+ # Collect results from successful submissions
345
+ timeout = self._backend_timeout
346
+ for i, handle in handles:
347
+ evaluation = self.evaluations[i]
348
+ eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
349
+ eval_version = getattr(evaluation, "version", "1.0.0")
350
+ try:
351
+ result = self._backend.get_result(handle, timeout=timeout)
352
+ results[i] = result
353
+ if callback:
354
+ callback(result)
355
+ except Exception as e:
356
+ error_result = EvalResult(
357
+ value=None,
358
+ eval_name=eval_name,
359
+ eval_version=eval_version,
360
+ latency_ms=0.0,
361
+ status=EvalStatus.FAILED,
362
+ error=str(e),
363
+ )
364
+ results[i] = error_result
365
+ if callback:
366
+ callback(error_result)
367
+
368
+ batch = BatchEvalResult.from_results(results)
369
+ return EvaluatorResult(
370
+ batch=batch,
371
+ mode=ExecutionMode.DISTRIBUTED,
372
+ )
373
+
374
+ @property
375
+ def _backend_timeout(self) -> float:
376
+ """Get timeout from backend config, defaulting to 300s."""
377
+ if self._backend and hasattr(self._backend, "config"):
378
+ config = self._backend.config
379
+ if hasattr(config, "timeout_seconds"):
380
+ return config.timeout_seconds
381
+ return 300.0
382
+
383
+ def _get_blocking_evaluator(self) -> BlockingEvaluator:
384
+ """Get or create the blocking evaluator."""
385
+ if self._blocking is None:
386
+ self._blocking = BlockingEvaluator(
387
+ evaluations=self.evaluations,
388
+ auto_enrich_span=self.auto_enrich_span,
389
+ fail_fast=self.fail_fast,
390
+ validate_inputs=self.validate_inputs,
391
+ )
392
+ return self._blocking
393
+
394
+ def _get_non_blocking_evaluator(self) -> NonBlockingEvaluator:
395
+ """Get or create the non-blocking evaluator."""
396
+ if self._non_blocking is None:
397
+ self._non_blocking = NonBlockingEvaluator(
398
+ evaluations=self.evaluations,
399
+ max_workers=self.max_workers,
400
+ auto_enrich_span=self.auto_enrich_span,
401
+ fail_fast=self.fail_fast,
402
+ validate_inputs=self.validate_inputs,
403
+ )
404
+ return self._non_blocking
405
+
406
+ def shutdown(self, wait: bool = True) -> None:
407
+ """
408
+ Shutdown the evaluator and release resources.
409
+
410
+ Args:
411
+ wait: Whether to wait for pending evaluations
412
+ """
413
+ if self._non_blocking:
414
+ self._non_blocking.shutdown(wait=wait)
415
+ self._non_blocking = None
416
+ if self._backend:
417
+ self._backend.shutdown(wait=wait)
418
+
419
+ def __enter__(self) -> "FrameworkEvaluator":
420
+ return self
421
+
422
+ def __exit__(self, exc_type, exc_val, exc_tb) -> None:
423
+ self.shutdown(wait=True)
424
+
425
+
426
+ # Factory functions for common configurations
427
+
428
+
429
+ def blocking_evaluator(
430
+ *evaluations: BaseEvaluation,
431
+ auto_enrich_span: bool = True,
432
+ fail_fast: bool = False,
433
+ ) -> FrameworkEvaluator:
434
+ """
435
+ Create a blocking evaluator.
436
+
437
+ Args:
438
+ *evaluations: Evaluations to run
439
+ auto_enrich_span: Whether to enrich OTEL spans
440
+ fail_fast: Stop on first failure
441
+
442
+ Returns:
443
+ Configured Evaluator in blocking mode
444
+
445
+ Example:
446
+ evaluator = blocking_evaluator(ToxicityEval(), BiasEval())
447
+ result = evaluator.run({"response": "..."})
448
+ """
449
+ return FrameworkEvaluator(
450
+ evaluations=list(evaluations),
451
+ mode=ExecutionMode.BLOCKING,
452
+ auto_enrich_span=auto_enrich_span,
453
+ fail_fast=fail_fast,
454
+ )
455
+
456
+
457
+ def async_evaluator(
458
+ *evaluations: BaseEvaluation,
459
+ max_workers: int = 4,
460
+ auto_enrich_span: bool = True,
461
+ backend: Optional[Backend] = None,
462
+ ) -> FrameworkEvaluator:
463
+ """
464
+ Create a non-blocking (async) evaluator.
465
+
466
+ Args:
467
+ *evaluations: Evaluations to run
468
+ max_workers: Maximum concurrent evaluations
469
+ auto_enrich_span: Whether to enrich OTEL spans
470
+ backend: Custom execution backend
471
+
472
+ Returns:
473
+ Configured Evaluator in non-blocking mode
474
+
475
+ Example:
476
+ evaluator = async_evaluator(ToxicityEval(), BiasEval())
477
+ result = evaluator.run({"response": "..."}) # Returns immediately
478
+ batch = result.wait() # Get results when needed
479
+ """
480
+ return FrameworkEvaluator(
481
+ evaluations=list(evaluations),
482
+ mode=ExecutionMode.NON_BLOCKING,
483
+ max_workers=max_workers,
484
+ auto_enrich_span=auto_enrich_span,
485
+ backend=backend,
486
+ )
487
+
488
+
489
+ def distributed_evaluator(
490
+ *evaluations: BaseEvaluation,
491
+ backend: Backend,
492
+ auto_enrich_span: bool = True,
493
+ ) -> FrameworkEvaluator:
494
+ """
495
+ Create a distributed evaluator with custom backend.
496
+
497
+ Args:
498
+ *evaluations: Evaluations to run
499
+ backend: Execution backend (Temporal, Celery, Ray, etc.)
500
+ auto_enrich_span: Whether to enrich OTEL spans
501
+
502
+ Returns:
503
+ Configured Evaluator in distributed mode
504
+
505
+ Example:
506
+ backend = TemporalBackend(config)
507
+ evaluator = distributed_evaluator(
508
+ ToxicityEval(),
509
+ backend=backend,
510
+ )
511
+ result = evaluator.run({"response": "..."})
512
+ """
513
+ return FrameworkEvaluator(
514
+ evaluations=list(evaluations),
515
+ mode=ExecutionMode.DISTRIBUTED,
516
+ backend=backend,
517
+ auto_enrich_span=auto_enrich_span,
518
+ )
519
+
520
+
521
+ def _execute_single_evaluation(
522
+ evaluation: BaseEvaluation,
523
+ inputs: Dict[str, Any],
524
+ validate: bool = True,
525
+ ) -> EvalResult:
526
+ """
527
+ Execute a single evaluation, suitable for submission to any backend.
528
+
529
+ Handles validation, timing, and error capture.
530
+
531
+ Args:
532
+ evaluation: The evaluation to run
533
+ inputs: Input data
534
+ validate: Whether to validate inputs
535
+
536
+ Returns:
537
+ EvalResult with value or error
538
+ """
539
+ eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
540
+ eval_version = getattr(evaluation, "version", "1.0.0")
541
+
542
+ # Validate inputs if enabled
543
+ if validate and hasattr(evaluation, "validate_inputs"):
544
+ try:
545
+ errors = evaluation.validate_inputs(inputs)
546
+ if errors:
547
+ error_msg = "; ".join(str(e) for e in errors) if isinstance(errors, list) else str(errors)
548
+ return EvalResult(
549
+ value=None,
550
+ eval_name=eval_name,
551
+ eval_version=eval_version,
552
+ latency_ms=0.0,
553
+ status=EvalStatus.FAILED,
554
+ error=f"Validation error: {error_msg}",
555
+ )
556
+ except Exception as e:
557
+ return EvalResult(
558
+ value=None,
559
+ eval_name=eval_name,
560
+ eval_version=eval_version,
561
+ latency_ms=0.0,
562
+ status=EvalStatus.FAILED,
563
+ error=f"Validation exception: {e}",
564
+ )
565
+
566
+ start = time.perf_counter()
567
+ try:
568
+ value = evaluation.evaluate(inputs)
569
+ latency_ms = (time.perf_counter() - start) * 1000
570
+ return EvalResult(
571
+ value=value,
572
+ eval_name=eval_name,
573
+ eval_version=eval_version,
574
+ latency_ms=latency_ms,
575
+ status=EvalStatus.COMPLETED,
576
+ )
577
+ except Exception as e:
578
+ latency_ms = (time.perf_counter() - start) * 1000
579
+ return EvalResult(
580
+ value=None,
581
+ eval_name=eval_name,
582
+ eval_version=eval_version,
583
+ latency_ms=latency_ms,
584
+ status=EvalStatus.FAILED,
585
+ error=str(e),
586
+ )
587
+
588
+
589
+ def resilient_evaluator(
590
+ *evaluations: BaseEvaluation,
591
+ backend: Optional[Backend] = None,
592
+ resilience: Optional[Any] = None,
593
+ fallback_backend: Optional[Backend] = None,
594
+ event_callback: Optional[Callable] = None,
595
+ auto_enrich_span: bool = True,
596
+ ) -> FrameworkEvaluator:
597
+ """
598
+ Create a distributed evaluator with resilience protections.
599
+
600
+ Wraps a backend with circuit breaker, rate limiting, retry, and fallback
601
+ capabilities, then creates an Evaluator in DISTRIBUTED mode.
602
+
603
+ Args:
604
+ *evaluations: Evaluations to run
605
+ backend: Execution backend (uses ThreadPoolBackend if None)
606
+ resilience: ResilienceConfig for resilience settings (optional)
607
+ fallback_backend: Optional fallback backend for degradation
608
+ event_callback: Callback for resilience events
609
+ auto_enrich_span: Whether to enrich OTEL spans
610
+
611
+ Returns:
612
+ Configured Evaluator in distributed mode with resilience
613
+
614
+ Example:
615
+ from fi.evals.framework import resilient_evaluator, ResilienceConfig
616
+ from fi.evals.framework import CircuitBreakerConfig, RateLimitConfig
617
+
618
+ evaluator = resilient_evaluator(
619
+ ToxicityEval(),
620
+ BiasEval(),
621
+ resilience=ResilienceConfig(
622
+ circuit_breaker=CircuitBreakerConfig(failure_threshold=5),
623
+ rate_limit=RateLimitConfig(requests_per_second=10),
624
+ ),
625
+ )
626
+ result = evaluator.run({"response": "..."})
627
+ """
628
+ from .resilience import ResilientBackend, ResilienceConfig
629
+
630
+ # Create underlying backend if not provided
631
+ underlying = backend or ThreadPoolBackend()
632
+
633
+ # Wrap with resilience
634
+ resilience_config = resilience or ResilienceConfig()
635
+ resilient = ResilientBackend(
636
+ underlying=underlying,
637
+ config=resilience_config,
638
+ fallback_backend=fallback_backend,
639
+ event_callback=event_callback,
640
+ )
641
+
642
+ return FrameworkEvaluator(
643
+ evaluations=list(evaluations),
644
+ mode=ExecutionMode.DISTRIBUTED,
645
+ backend=resilient,
646
+ auto_enrich_span=auto_enrich_span,
647
+ )
@@ -0,0 +1,22 @@
1
+ """Evaluator implementations for different execution modes."""
2
+
3
+ from .blocking import BlockingEvaluator, blocking_evaluate
4
+ from .non_blocking import (
5
+ NonBlockingEvaluator,
6
+ non_blocking_evaluate,
7
+ EvalFuture,
8
+ BatchEvalFuture,
9
+ EvalResultAggregator,
10
+ )
11
+
12
+ __all__ = [
13
+ # Blocking
14
+ "BlockingEvaluator",
15
+ "blocking_evaluate",
16
+ # Non-blocking
17
+ "NonBlockingEvaluator",
18
+ "non_blocking_evaluate",
19
+ "EvalFuture",
20
+ "BatchEvalFuture",
21
+ "EvalResultAggregator",
22
+ ]