agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,690 @@
1
+ """Local evaluator for running metrics without API calls.
2
+
3
+ This module provides the LocalEvaluator class which can run heuristic
4
+ metrics locally, enabling offline evaluation and faster feedback loops.
5
+
6
+ It also provides the HybridEvaluator which can intelligently route
7
+ evaluations between local execution, local LLM, and cloud APIs.
8
+ """
9
+
10
+ from dataclasses import dataclass, field
11
+ from typing import Any, Dict, List, Optional, TYPE_CHECKING
12
+ import time
13
+ import logging
14
+
15
+ from ..types import BatchRunResult, EvalResult
16
+ from .execution_mode import RoutingMode, can_run_locally
17
+ from .registry import get_registry, LocalMetricRegistry
18
+
19
+ if TYPE_CHECKING:
20
+ from .llm import OllamaLLM
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+
25
+ @dataclass
26
+ class LocalEvaluatorConfig:
27
+ """Configuration for the local evaluator.
28
+
29
+ Attributes:
30
+ execution_mode: The default execution mode.
31
+ fail_on_unsupported: If True, raise an error when a metric can't run locally.
32
+ parallel_workers: Number of parallel workers (for future use).
33
+ timeout: Timeout in seconds for individual evaluations.
34
+ """
35
+
36
+ execution_mode: RoutingMode = RoutingMode.HYBRID
37
+ fail_on_unsupported: bool = False
38
+ parallel_workers: int = 4
39
+ timeout: int = 60
40
+
41
+
42
+ @dataclass
43
+ class LocalEvaluationResult:
44
+ """Result of a local evaluation operation.
45
+
46
+ Attributes:
47
+ results: The batch run results.
48
+ executed_locally: Set of metric names that ran locally.
49
+ skipped: Set of metric names that were skipped (not local-capable).
50
+ errors: Dictionary mapping metric names to error messages.
51
+ """
52
+
53
+ results: BatchRunResult
54
+ executed_locally: set = field(default_factory=set)
55
+ skipped: set = field(default_factory=set)
56
+ errors: Dict[str, str] = field(default_factory=dict)
57
+
58
+
59
+ class LocalEvaluator:
60
+ """Evaluator that runs metrics locally without API calls.
61
+
62
+ This evaluator can run heuristic metrics locally, providing fast feedback
63
+ without requiring network access or API credentials.
64
+
65
+ Example:
66
+ >>> evaluator = LocalEvaluator()
67
+ >>> result = evaluator.evaluate(
68
+ ... metric_name="contains",
69
+ ... inputs=[{"response": "Hello world", "keyword": "world"}],
70
+ ... config={"keyword": "world"}
71
+ ... )
72
+ >>> print(result.results.eval_results[0].output)
73
+ 1.0
74
+ """
75
+
76
+ def __init__(
77
+ self,
78
+ config: Optional[LocalEvaluatorConfig] = None,
79
+ registry: Optional[LocalMetricRegistry] = None,
80
+ ) -> None:
81
+ """Initialize the local evaluator.
82
+
83
+ Args:
84
+ config: Configuration for the evaluator.
85
+ registry: Optional metric registry (uses global if not provided).
86
+ """
87
+ self.config = config or LocalEvaluatorConfig()
88
+ self.registry = registry or get_registry()
89
+
90
+ def can_run_locally(self, metric_name: str) -> bool:
91
+ """Check if a metric can be run locally.
92
+
93
+ Args:
94
+ metric_name: The name of the metric.
95
+
96
+ Returns:
97
+ True if the metric can run locally.
98
+ """
99
+ return can_run_locally(metric_name) and self.registry.is_registered(metric_name)
100
+
101
+ def evaluate(
102
+ self,
103
+ metric_name: str,
104
+ inputs: List[Dict[str, Any]],
105
+ config: Optional[Dict[str, Any]] = None,
106
+ ) -> LocalEvaluationResult:
107
+ """Evaluate a single metric on a batch of inputs.
108
+
109
+ Args:
110
+ metric_name: The name of the metric to run.
111
+ inputs: List of input dictionaries for the metric.
112
+ config: Optional configuration for the metric.
113
+
114
+ Returns:
115
+ LocalEvaluationResult with results and metadata.
116
+
117
+ Raises:
118
+ ValueError: If fail_on_unsupported is True and metric can't run locally.
119
+ """
120
+ result = LocalEvaluationResult(results=BatchRunResult(eval_results=[]))
121
+
122
+ if not self.can_run_locally(metric_name):
123
+ if self.config.fail_on_unsupported:
124
+ raise ValueError(
125
+ f"Metric '{metric_name}' cannot run locally. "
126
+ f"Available local metrics: {self.registry.list_metrics()}"
127
+ )
128
+ result.skipped.add(metric_name)
129
+ # Return empty results for skipped metrics
130
+ for _ in inputs:
131
+ result.results.eval_results.append(
132
+ EvalResult(
133
+ name=metric_name,
134
+ output=None,
135
+ reason=f"Metric '{metric_name}' cannot run locally",
136
+ runtime=0,
137
+ )
138
+ )
139
+ return result
140
+
141
+ try:
142
+ metric = self.registry.create(metric_name, config)
143
+ if metric is None:
144
+ raise ValueError(f"Failed to create metric '{metric_name}'")
145
+
146
+ batch_result = metric.evaluate(inputs)
147
+ result.results = batch_result
148
+ result.executed_locally.add(metric_name)
149
+
150
+ except Exception as e:
151
+ result.errors[metric_name] = str(e)
152
+ # Fill with error results
153
+ for _ in inputs:
154
+ result.results.eval_results.append(
155
+ EvalResult(
156
+ name=metric_name,
157
+ output=None,
158
+ reason=f"Error: {str(e)}",
159
+ runtime=0,
160
+ )
161
+ )
162
+
163
+ # Automatically enrich current span with evaluation results
164
+ try:
165
+ from fi.evals.otel.enrichment import enrich_span_with_batch_result, is_auto_enrichment_enabled
166
+ if is_auto_enrichment_enabled():
167
+ enriched_count = enrich_span_with_batch_result(result.results)
168
+ if enriched_count > 0:
169
+ import logging
170
+ logging.debug(f"Enriched active span with {enriched_count} evaluation results")
171
+ except ImportError:
172
+ pass # OTEL enrichment not available
173
+ except Exception:
174
+ pass # Silently fail enrichment
175
+
176
+ return result
177
+
178
+ def evaluate_batch(
179
+ self,
180
+ evaluations: List[Dict[str, Any]],
181
+ ) -> LocalEvaluationResult:
182
+ """Evaluate multiple metrics on their respective inputs.
183
+
184
+ Args:
185
+ evaluations: List of evaluation specifications, each containing:
186
+ - metric_name: Name of the metric
187
+ - inputs: List of input dictionaries
188
+ - config: Optional metric configuration
189
+
190
+ Returns:
191
+ LocalEvaluationResult with combined results.
192
+
193
+ Example:
194
+ >>> evaluator = LocalEvaluator()
195
+ >>> result = evaluator.evaluate_batch([
196
+ ... {
197
+ ... "metric_name": "contains",
198
+ ... "inputs": [{"response": "hello world"}],
199
+ ... "config": {"keyword": "world"}
200
+ ... },
201
+ ... {
202
+ ... "metric_name": "is_json",
203
+ ... "inputs": [{"response": '{"key": "value"}'}]
204
+ ... }
205
+ ... ])
206
+ """
207
+ combined_result = LocalEvaluationResult(
208
+ results=BatchRunResult(eval_results=[])
209
+ )
210
+
211
+ for eval_spec in evaluations:
212
+ metric_name = eval_spec.get("metric_name")
213
+ inputs = eval_spec.get("inputs", [])
214
+ config = eval_spec.get("config")
215
+
216
+ if not metric_name:
217
+ continue
218
+
219
+ single_result = self.evaluate(metric_name, inputs, config)
220
+
221
+ # Merge results
222
+ combined_result.results.eval_results.extend(
223
+ single_result.results.eval_results
224
+ )
225
+ combined_result.executed_locally.update(single_result.executed_locally)
226
+ combined_result.skipped.update(single_result.skipped)
227
+ combined_result.errors.update(single_result.errors)
228
+
229
+ return combined_result
230
+
231
+ def list_available_metrics(self) -> List[str]:
232
+ """List all metrics available for local execution.
233
+
234
+ Returns:
235
+ Sorted list of available metric names.
236
+ """
237
+ return self.registry.list_metrics()
238
+
239
+
240
+ class HybridEvaluator:
241
+ """Evaluator that routes metrics between local and cloud execution.
242
+
243
+ This evaluator analyzes each metric and automatically routes it to
244
+ either local heuristics, local LLM, or cloud execution based on the
245
+ metric type and configuration.
246
+
247
+ Example:
248
+ >>> # Basic usage with automatic routing
249
+ >>> evaluator = HybridEvaluator()
250
+ >>> partitions = evaluator.partition_evaluations([
251
+ ... {"metric_name": "contains", "inputs": [{"response": "test"}]},
252
+ ... {"metric_name": "groundedness", "inputs": [{"response": "test"}]},
253
+ ... ])
254
+ >>> local_results = evaluator.evaluate_local_partition(partitions[RoutingMode.LOCAL])
255
+
256
+ >>> # With local LLM for LLM-based evaluations
257
+ >>> from fi.evals.local.llm import OllamaLLM
258
+ >>> evaluator = HybridEvaluator(local_llm=OllamaLLM())
259
+ >>> result = evaluator.evaluate(
260
+ ... template="custom_llm_judge",
261
+ ... inputs=[{"query": "What is AI?", "response": "AI is..."}]
262
+ ... )
263
+ """
264
+
265
+ # LLM-based metrics that can use local LLM instead of cloud
266
+ LLM_BASED_METRICS = {
267
+ "groundedness",
268
+ "hallucination",
269
+ "relevance",
270
+ "coherence",
271
+ "context_relevance",
272
+ "answer_relevance",
273
+ "custom_llm_judge",
274
+ "tone",
275
+ "safety",
276
+ "pii",
277
+ "bias",
278
+ }
279
+
280
+ def __init__(
281
+ self,
282
+ config: Optional[LocalEvaluatorConfig] = None,
283
+ local_evaluator: Optional[LocalEvaluator] = None,
284
+ local_llm: Optional["OllamaLLM"] = None,
285
+ cloud_evaluator: Optional[Any] = None,
286
+ prefer_local: bool = True,
287
+ fallback_to_cloud: bool = True,
288
+ offline_mode: bool = False,
289
+ ) -> None:
290
+ """Initialize the hybrid evaluator.
291
+
292
+ Args:
293
+ config: Configuration for the evaluator.
294
+ local_evaluator: Optional local evaluator instance for heuristic metrics.
295
+ local_llm: Optional local LLM instance for LLM-based metrics.
296
+ cloud_evaluator: Optional cloud evaluator instance (Evaluator).
297
+ prefer_local: If True, prefer local execution when possible.
298
+ fallback_to_cloud: If True, fall back to cloud when local fails.
299
+ offline_mode: If True, never use cloud (raise error if metric requires it).
300
+ """
301
+ self.config = config or LocalEvaluatorConfig(execution_mode=RoutingMode.HYBRID)
302
+ self.local_evaluator = local_evaluator or LocalEvaluator(self.config)
303
+ self.local_llm = local_llm
304
+ self.cloud_evaluator = cloud_evaluator
305
+ self.prefer_local = prefer_local
306
+ self.fallback_to_cloud = fallback_to_cloud
307
+ self.offline_mode = offline_mode
308
+
309
+ def set_local_llm(self, llm: "OllamaLLM") -> None:
310
+ """Set the local LLM instance.
311
+
312
+ Args:
313
+ llm: OllamaLLM instance to use for LLM-based evaluations.
314
+ """
315
+ self.local_llm = llm
316
+
317
+ def set_cloud_evaluator(self, evaluator: Any) -> None:
318
+ """Set the cloud evaluator instance.
319
+
320
+ Args:
321
+ evaluator: Cloud Evaluator instance for API-based evaluations.
322
+ """
323
+ self.cloud_evaluator = evaluator
324
+
325
+ def can_use_local_llm(self, metric_name: str) -> bool:
326
+ """Check if a metric can use the local LLM.
327
+
328
+ Args:
329
+ metric_name: The name of the metric.
330
+
331
+ Returns:
332
+ True if the metric can use local LLM.
333
+ """
334
+ if self.local_llm is None:
335
+ return False
336
+ if not self.local_llm.is_available():
337
+ return False
338
+ return metric_name.lower() in self.LLM_BASED_METRICS
339
+
340
+ def route_evaluation(
341
+ self,
342
+ metric_name: str,
343
+ force_local: bool = False,
344
+ force_cloud: bool = False,
345
+ ) -> RoutingMode:
346
+ """Determine the execution mode for a metric.
347
+
348
+ Args:
349
+ metric_name: The name of the metric.
350
+ force_local: Force local execution.
351
+ force_cloud: Force cloud execution.
352
+
353
+ Returns:
354
+ The recommended execution mode.
355
+ """
356
+ if force_cloud and not self.offline_mode:
357
+ return RoutingMode.CLOUD
358
+ if force_local:
359
+ return RoutingMode.LOCAL
360
+
361
+ # Check if it's a heuristic metric that can run locally
362
+ if can_run_locally(metric_name):
363
+ return RoutingMode.LOCAL
364
+
365
+ # Check if it's an LLM metric and we have local LLM
366
+ if self.can_use_local_llm(metric_name) and self.prefer_local:
367
+ return RoutingMode.LOCAL
368
+
369
+ # Default to cloud unless in offline mode
370
+ if self.offline_mode:
371
+ raise ValueError(
372
+ f"Metric '{metric_name}' requires cloud execution but offline_mode is enabled"
373
+ )
374
+ return RoutingMode.CLOUD
375
+
376
+ def partition_evaluations(
377
+ self, evaluations: List[Dict[str, Any]]
378
+ ) -> Dict[RoutingMode, List[Dict[str, Any]]]:
379
+ """Partition evaluations by execution mode.
380
+
381
+ Args:
382
+ evaluations: List of evaluation specifications.
383
+
384
+ Returns:
385
+ Dictionary mapping execution modes to their evaluations.
386
+ """
387
+ partitions: Dict[RoutingMode, List[Dict[str, Any]]] = {
388
+ RoutingMode.LOCAL: [],
389
+ RoutingMode.CLOUD: [],
390
+ }
391
+
392
+ for eval_spec in evaluations:
393
+ metric_name = eval_spec.get("metric_name", "")
394
+ force_local = eval_spec.get("force_local", False)
395
+ force_cloud = eval_spec.get("force_cloud", False)
396
+
397
+ try:
398
+ mode = self.route_evaluation(metric_name, force_local, force_cloud)
399
+ partitions[mode].append(eval_spec)
400
+ except ValueError as e:
401
+ # Offline mode violation - add to errors
402
+ logger.error(f"Routing error for {metric_name}: {e}")
403
+ partitions[RoutingMode.CLOUD].append(eval_spec)
404
+
405
+ return partitions
406
+
407
+ def evaluate_local_partition(
408
+ self, evaluations: List[Dict[str, Any]]
409
+ ) -> LocalEvaluationResult:
410
+ """Evaluate the local partition of evaluations.
411
+
412
+ This handles both heuristic metrics (via LocalEvaluator) and
413
+ LLM-based metrics (via local LLM).
414
+
415
+ Args:
416
+ evaluations: List of evaluation specifications to run locally.
417
+
418
+ Returns:
419
+ LocalEvaluationResult with results.
420
+ """
421
+ heuristic_evals = []
422
+ llm_evals = []
423
+
424
+ # Separate heuristic and LLM evaluations
425
+ for eval_spec in evaluations:
426
+ metric_name = eval_spec.get("metric_name", "")
427
+ if can_run_locally(metric_name):
428
+ heuristic_evals.append(eval_spec)
429
+ elif self.can_use_local_llm(metric_name):
430
+ llm_evals.append(eval_spec)
431
+ else:
432
+ heuristic_evals.append(eval_spec) # Will be skipped
433
+
434
+ # Run heuristic evaluations
435
+ result = self.local_evaluator.evaluate_batch(heuristic_evals)
436
+
437
+ # Run LLM evaluations if we have a local LLM
438
+ if llm_evals and self.local_llm:
439
+ llm_results = self._evaluate_with_local_llm(llm_evals)
440
+ result.results.eval_results.extend(llm_results.results.eval_results)
441
+ result.executed_locally.update(llm_results.executed_locally)
442
+ result.skipped.update(llm_results.skipped)
443
+ result.errors.update(llm_results.errors)
444
+
445
+ return result
446
+
447
+ def _evaluate_with_local_llm(
448
+ self, evaluations: List[Dict[str, Any]]
449
+ ) -> LocalEvaluationResult:
450
+ """Run evaluations using the local LLM.
451
+
452
+ Args:
453
+ evaluations: List of LLM-based evaluation specifications.
454
+
455
+ Returns:
456
+ LocalEvaluationResult with LLM evaluation results.
457
+ """
458
+ result = LocalEvaluationResult(results=BatchRunResult(eval_results=[]))
459
+
460
+ if not self.local_llm:
461
+ for eval_spec in evaluations:
462
+ metric_name = eval_spec.get("metric_name", "")
463
+ result.skipped.add(metric_name)
464
+ for _ in eval_spec.get("inputs", []):
465
+ result.results.eval_results.append(
466
+ EvalResult(
467
+ name=metric_name,
468
+ output=None,
469
+ reason="No local LLM configured",
470
+ runtime=0,
471
+ )
472
+ )
473
+ return result
474
+
475
+ for eval_spec in evaluations:
476
+ metric_name = eval_spec.get("metric_name", "")
477
+ inputs = eval_spec.get("inputs", [])
478
+ config = eval_spec.get("config", {})
479
+
480
+ for input_data in inputs:
481
+ start_time = time.time()
482
+ try:
483
+ # Build evaluation from input
484
+ judge_result = self.local_llm.judge(
485
+ query=input_data.get("input", input_data.get("query", "")),
486
+ response=input_data.get("response", input_data.get("output", "")),
487
+ criteria=config.get("criteria", f"Evaluate based on {metric_name}"),
488
+ context=input_data.get("context", input_data.get("contexts", "")),
489
+ )
490
+
491
+ runtime = int((time.time() - start_time) * 1000)
492
+ result.results.eval_results.append(
493
+ EvalResult(
494
+ name=metric_name,
495
+ output=judge_result.get("score", 0.0),
496
+ reason=judge_result.get("reason", ""),
497
+ runtime=runtime,
498
+ metrics=[{
499
+ "name": metric_name,
500
+ "value": judge_result.get("score", 0.0),
501
+ }],
502
+ )
503
+ )
504
+ result.executed_locally.add(metric_name)
505
+
506
+ except Exception as e:
507
+ runtime = int((time.time() - start_time) * 1000)
508
+ result.errors[metric_name] = str(e)
509
+ result.results.eval_results.append(
510
+ EvalResult(
511
+ name=metric_name,
512
+ output=0.0,
513
+ reason=f"Local LLM error: {str(e)}",
514
+ runtime=runtime,
515
+ )
516
+ )
517
+
518
+ return result
519
+
520
+ def evaluate(
521
+ self,
522
+ template: str,
523
+ inputs: List[Dict[str, Any]],
524
+ config: Optional[Dict[str, Any]] = None,
525
+ ) -> LocalEvaluationResult:
526
+ """Evaluate a single template with automatic routing.
527
+
528
+ Args:
529
+ template: The evaluation template/metric name.
530
+ inputs: List of input dictionaries.
531
+ config: Optional configuration for the evaluation.
532
+
533
+ Returns:
534
+ LocalEvaluationResult with evaluation results.
535
+ """
536
+ eval_spec = {
537
+ "metric_name": template,
538
+ "inputs": inputs,
539
+ "config": config or {},
540
+ }
541
+
542
+ mode = self.route_evaluation(template)
543
+
544
+ if mode == RoutingMode.LOCAL:
545
+ result = self.evaluate_local_partition([eval_spec])
546
+ else:
547
+ result = self._evaluate_cloud([eval_spec])
548
+
549
+ # Automatically enrich current span with evaluation results
550
+ try:
551
+ from fi.evals.otel.enrichment import enrich_span_with_batch_result, is_auto_enrichment_enabled
552
+ if is_auto_enrichment_enabled():
553
+ enrich_span_with_batch_result(result.results)
554
+ except ImportError:
555
+ pass
556
+ except Exception:
557
+ pass
558
+
559
+ return result
560
+
561
+ def evaluate_batch(
562
+ self,
563
+ evaluations: List[Dict[str, Any]],
564
+ ) -> LocalEvaluationResult:
565
+ """Evaluate multiple templates with automatic routing.
566
+
567
+ Args:
568
+ evaluations: List of evaluation specifications.
569
+
570
+ Returns:
571
+ Combined LocalEvaluationResult from all evaluations.
572
+ """
573
+ partitions = self.partition_evaluations(evaluations)
574
+
575
+ # Run local evaluations
576
+ local_result = self.evaluate_local_partition(partitions[RoutingMode.LOCAL])
577
+
578
+ # Run cloud evaluations
579
+ if partitions[RoutingMode.CLOUD]:
580
+ cloud_result = self._evaluate_cloud(partitions[RoutingMode.CLOUD])
581
+
582
+ # Merge results
583
+ local_result.results.eval_results.extend(cloud_result.results.eval_results)
584
+ local_result.executed_locally.update(cloud_result.executed_locally)
585
+ local_result.skipped.update(cloud_result.skipped)
586
+ local_result.errors.update(cloud_result.errors)
587
+
588
+ return local_result
589
+
590
+ def _evaluate_cloud(
591
+ self, evaluations: List[Dict[str, Any]]
592
+ ) -> LocalEvaluationResult:
593
+ """Run evaluations via cloud API.
594
+
595
+ Args:
596
+ evaluations: List of evaluation specifications for cloud.
597
+
598
+ Returns:
599
+ LocalEvaluationResult with cloud evaluation results.
600
+ """
601
+ result = LocalEvaluationResult(results=BatchRunResult(eval_results=[]))
602
+
603
+ if self.offline_mode:
604
+ for eval_spec in evaluations:
605
+ metric_name = eval_spec.get("metric_name", "")
606
+ result.skipped.add(metric_name)
607
+ for _ in eval_spec.get("inputs", []):
608
+ result.results.eval_results.append(
609
+ EvalResult(
610
+ name=metric_name,
611
+ output=None,
612
+ reason="Offline mode - cloud execution disabled",
613
+ runtime=0,
614
+ )
615
+ )
616
+ return result
617
+
618
+ if not self.cloud_evaluator:
619
+ for eval_spec in evaluations:
620
+ metric_name = eval_spec.get("metric_name", "")
621
+ result.skipped.add(metric_name)
622
+ for _ in eval_spec.get("inputs", []):
623
+ result.results.eval_results.append(
624
+ EvalResult(
625
+ name=metric_name,
626
+ output=None,
627
+ reason="No cloud evaluator configured",
628
+ runtime=0,
629
+ )
630
+ )
631
+ return result
632
+
633
+ # Route through fi.evals.evaluate() with Turing engine
634
+ try:
635
+ from fi.evals import evaluate as core_evaluate
636
+
637
+ for eval_spec in evaluations:
638
+ metric_name = eval_spec.get("metric_name", "")
639
+ inputs_list = eval_spec.get("inputs", [])
640
+
641
+ for input_data in inputs_list:
642
+ start_time = time.time()
643
+ try:
644
+ eval_result = core_evaluate(
645
+ metric_name,
646
+ engine="turing",
647
+ output=input_data.get("response", input_data.get("output", "")),
648
+ input=input_data.get("input", input_data.get("query", "")),
649
+ context=input_data.get("context", input_data.get("contexts", "")),
650
+ )
651
+
652
+ runtime = int((time.time() - start_time) * 1000)
653
+ result.results.eval_results.append(
654
+ EvalResult(
655
+ name=metric_name,
656
+ output=eval_result.score if hasattr(eval_result, "score") else eval_result.output,
657
+ reason=getattr(eval_result, "reason", ""),
658
+ runtime=runtime,
659
+ metrics=[{
660
+ "name": metric_name,
661
+ "value": eval_result.score if hasattr(eval_result, "score") else 0.0,
662
+ }],
663
+ )
664
+ )
665
+ result.executed_locally.add(metric_name)
666
+
667
+ except Exception as e:
668
+ runtime = int((time.time() - start_time) * 1000)
669
+ result.errors[metric_name] = str(e)
670
+ result.results.eval_results.append(
671
+ EvalResult(
672
+ name=metric_name,
673
+ output=None,
674
+ reason=f"Cloud evaluation error: {e}",
675
+ runtime=runtime,
676
+ )
677
+ )
678
+
679
+ except ImportError:
680
+ logger.error("fi.evals.evaluate not available for cloud routing")
681
+ for eval_spec in evaluations:
682
+ metric_name = eval_spec.get("metric_name", "")
683
+ result.skipped.add(metric_name)
684
+ result.errors[metric_name] = "fi.evals.evaluate not available"
685
+ except Exception as e:
686
+ logger.error(f"Cloud evaluation failed: {e}")
687
+ for eval_spec in evaluations:
688
+ result.errors[eval_spec.get("metric_name", "")] = str(e)
689
+
690
+ return result