agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,121 @@
1
+ """Execution mode selector for evaluations.
2
+
3
+ This module defines execution modes that determine how evaluations run:
4
+ - LOCAL: Run all evaluations locally using heuristic metrics (no API calls)
5
+ - CLOUD: Run all evaluations via the cloud API
6
+ - HYBRID: Automatically route each evaluation to local or cloud based on metric type
7
+ """
8
+
9
+ from enum import Enum
10
+ from typing import Set
11
+
12
+
13
+ class RoutingMode(Enum):
14
+ """Defines how evaluations should be executed."""
15
+
16
+ LOCAL = "local"
17
+ """Run evaluations locally using heuristic metrics only."""
18
+
19
+ CLOUD = "cloud"
20
+ """Run all evaluations via the cloud API."""
21
+
22
+ HYBRID = "hybrid"
23
+ """Automatically choose local or cloud based on metric capabilities."""
24
+
25
+ def __str__(self) -> str:
26
+ return self.value
27
+
28
+
29
+ # Set of metric names that can be run locally (heuristic metrics)
30
+ LOCAL_CAPABLE_METRICS: Set[str] = {
31
+ # String metrics
32
+ "regex",
33
+ "contains",
34
+ "contains_all",
35
+ "contains_any",
36
+ "contains_none",
37
+ "one_line",
38
+ "contains_email",
39
+ "is_email",
40
+ "contains_link",
41
+ "contains_valid_link",
42
+ "equals",
43
+ "starts_with",
44
+ "ends_with",
45
+ "length_less_than",
46
+ "length_greater_than",
47
+ "length_between",
48
+
49
+ # JSON metrics
50
+ "contains_json",
51
+ "is_json",
52
+ "json_schema",
53
+
54
+ # Similarity metrics
55
+ "bleu_score",
56
+ "rouge_score",
57
+ "recall_score",
58
+ "levenshtein_similarity",
59
+ "numeric_similarity",
60
+ "embedding_similarity",
61
+ "semantic_list_contains",
62
+ }
63
+
64
+
65
+ def can_run_locally(metric_name: str) -> bool:
66
+ """Check if a metric can be run locally.
67
+
68
+ Args:
69
+ metric_name: The name of the metric to check.
70
+
71
+ Returns:
72
+ True if the metric can run locally, False otherwise.
73
+ """
74
+ return metric_name.lower() in LOCAL_CAPABLE_METRICS
75
+
76
+
77
+ def select_routing_mode(
78
+ metric_name: str,
79
+ preferred_mode: RoutingMode,
80
+ force_local: bool = False,
81
+ force_cloud: bool = False,
82
+ ) -> RoutingMode:
83
+ """Select the execution mode for a metric based on preferences and capabilities.
84
+
85
+ Args:
86
+ metric_name: The name of the metric.
87
+ preferred_mode: The user's preferred execution mode.
88
+ force_local: If True, always try local execution (error if not possible).
89
+ force_cloud: If True, always use cloud execution.
90
+
91
+ Returns:
92
+ The selected execution mode.
93
+
94
+ Raises:
95
+ ValueError: If force_local is True but the metric cannot run locally.
96
+ """
97
+ if force_cloud:
98
+ return RoutingMode.CLOUD
99
+
100
+ if force_local:
101
+ if not can_run_locally(metric_name):
102
+ raise ValueError(
103
+ f"Metric '{metric_name}' cannot run locally. "
104
+ f"Local-capable metrics: {sorted(LOCAL_CAPABLE_METRICS)}"
105
+ )
106
+ return RoutingMode.LOCAL
107
+
108
+ if preferred_mode == RoutingMode.LOCAL:
109
+ if can_run_locally(metric_name):
110
+ return RoutingMode.LOCAL
111
+ # Fall back to cloud if metric can't run locally
112
+ return RoutingMode.CLOUD
113
+
114
+ if preferred_mode == RoutingMode.HYBRID:
115
+ # In hybrid mode, prefer local for capable metrics
116
+ if can_run_locally(metric_name):
117
+ return RoutingMode.LOCAL
118
+ return RoutingMode.CLOUD
119
+
120
+ # Default: CLOUD mode
121
+ return RoutingMode.CLOUD
fi/evals/local/llm.py ADDED
@@ -0,0 +1,489 @@
1
+ """Local LLM integration for running LLM-as-judge evaluations without cloud API.
2
+
3
+ This module provides support for local LLM inference via Ollama, enabling
4
+ offline LLM-based evaluations for air-gapped environments or faster iteration.
5
+
6
+ Example:
7
+ >>> from fi.evals.local.llm import OllamaLLM, LocalLLMConfig
8
+ >>>
9
+ >>> # Initialize with default model
10
+ >>> llm = OllamaLLM()
11
+ >>>
12
+ >>> # Generate completion
13
+ >>> response = llm.generate("What is 2+2?")
14
+ >>>
15
+ >>> # Use as LLM judge
16
+ >>> result = llm.judge(
17
+ ... query="What is the capital of France?",
18
+ ... response="The capital of France is Paris.",
19
+ ... criteria="Evaluate if the response correctly answers the question."
20
+ ... )
21
+ >>> print(result["score"])
22
+ 0.9
23
+ """
24
+
25
+ from dataclasses import dataclass
26
+ from typing import Any, Dict, List, Optional
27
+ import json
28
+ import logging
29
+ import re
30
+
31
+ logger = logging.getLogger(__name__)
32
+
33
+
34
+ @dataclass
35
+ class LocalLLMConfig:
36
+ """Configuration for local LLM.
37
+
38
+ Attributes:
39
+ model: The model name to use (e.g., "llama3.2", "mistral", "phi3").
40
+ base_url: The Ollama API base URL.
41
+ temperature: Sampling temperature (0.0 for deterministic).
42
+ max_tokens: Maximum tokens to generate.
43
+ timeout: Request timeout in seconds.
44
+ """
45
+
46
+ model: str = "llama3.2"
47
+ base_url: str = "http://localhost:11434"
48
+ temperature: float = 0.0
49
+ max_tokens: int = 1024
50
+ timeout: int = 120
51
+
52
+
53
+ class OllamaLLM:
54
+ """Interface to Ollama for local LLM inference.
55
+
56
+ This class provides methods for text generation and LLM-as-judge
57
+ evaluations using locally running Ollama models.
58
+
59
+ Example:
60
+ >>> llm = OllamaLLM()
61
+ >>> llm.is_available()
62
+ True
63
+ >>> response = llm.generate("Hello, how are you?")
64
+ "I'm doing well, thank you for asking!"
65
+ """
66
+
67
+ def __init__(
68
+ self,
69
+ config: Optional[LocalLLMConfig] = None,
70
+ auto_check: bool = True,
71
+ ) -> None:
72
+ """Initialize the Ollama LLM interface.
73
+
74
+ Args:
75
+ config: Configuration for the LLM.
76
+ auto_check: If True, check Ollama availability on init.
77
+ """
78
+ self.config = config or LocalLLMConfig()
79
+ self._available: Optional[bool] = None
80
+ self._models: Optional[List[str]] = None
81
+
82
+ if auto_check:
83
+ self._check_availability()
84
+
85
+ def _check_availability(self) -> None:
86
+ """Check if Ollama is available and the model is installed."""
87
+ try:
88
+ import requests
89
+
90
+ response = requests.get(
91
+ f"{self.config.base_url}/api/tags",
92
+ timeout=5
93
+ )
94
+ response.raise_for_status()
95
+
96
+ data = response.json()
97
+ self._models = [m.get("name", "") for m in data.get("models", [])]
98
+ self._available = True
99
+
100
+ # Check if requested model is available
101
+ model_base = self.config.model.split(":")[0]
102
+ if not any(model_base in m for m in self._models):
103
+ logger.warning(
104
+ f"Model '{self.config.model}' not found in Ollama. "
105
+ f"Available models: {self._models}"
106
+ )
107
+
108
+ except Exception as e:
109
+ self._available = False
110
+ logger.warning(f"Ollama not available: {e}")
111
+
112
+ def is_available(self) -> bool:
113
+ """Check if Ollama is available.
114
+
115
+ Returns:
116
+ True if Ollama is running and accessible.
117
+ """
118
+ if self._available is None:
119
+ self._check_availability()
120
+ return self._available or False
121
+
122
+ def list_models(self) -> List[str]:
123
+ """List available models in Ollama.
124
+
125
+ Returns:
126
+ List of installed model names.
127
+ """
128
+ if self._models is None:
129
+ self._check_availability()
130
+ return self._models or []
131
+
132
+ def generate(
133
+ self,
134
+ prompt: str,
135
+ system: Optional[str] = None,
136
+ temperature: Optional[float] = None,
137
+ max_tokens: Optional[int] = None,
138
+ ) -> str:
139
+ """Generate a completion using Ollama.
140
+
141
+ Args:
142
+ prompt: The user prompt.
143
+ system: Optional system prompt.
144
+ temperature: Override temperature for this request.
145
+ max_tokens: Override max tokens for this request.
146
+
147
+ Returns:
148
+ The generated text response.
149
+
150
+ Raises:
151
+ ConnectionError: If Ollama is not available.
152
+ RuntimeError: If generation fails.
153
+ """
154
+ if not self.is_available():
155
+ raise ConnectionError(
156
+ f"Cannot connect to Ollama at {self.config.base_url}. "
157
+ "Make sure Ollama is running: `ollama serve`"
158
+ )
159
+
160
+ import requests
161
+
162
+ payload = {
163
+ "model": self.config.model,
164
+ "prompt": prompt,
165
+ "stream": False,
166
+ "options": {
167
+ "temperature": temperature if temperature is not None else self.config.temperature,
168
+ "num_predict": max_tokens if max_tokens is not None else self.config.max_tokens,
169
+ }
170
+ }
171
+
172
+ if system:
173
+ payload["system"] = system
174
+
175
+ try:
176
+ response = requests.post(
177
+ f"{self.config.base_url}/api/generate",
178
+ json=payload,
179
+ timeout=self.config.timeout,
180
+ )
181
+ response.raise_for_status()
182
+
183
+ return response.json().get("response", "")
184
+
185
+ except requests.exceptions.Timeout:
186
+ raise RuntimeError(
187
+ f"Ollama request timed out after {self.config.timeout}s"
188
+ )
189
+ except requests.exceptions.RequestException as e:
190
+ raise RuntimeError(f"Ollama request failed: {e}")
191
+
192
+ def chat(
193
+ self,
194
+ messages: List[Dict[str, str]],
195
+ temperature: Optional[float] = None,
196
+ max_tokens: Optional[int] = None,
197
+ ) -> str:
198
+ """Chat completion using Ollama's chat API.
199
+
200
+ Args:
201
+ messages: List of message dictionaries with 'role' and 'content'.
202
+ temperature: Override temperature for this request.
203
+ max_tokens: Override max tokens for this request.
204
+
205
+ Returns:
206
+ The assistant's response.
207
+ """
208
+ if not self.is_available():
209
+ raise ConnectionError(
210
+ f"Cannot connect to Ollama at {self.config.base_url}. "
211
+ "Make sure Ollama is running: `ollama serve`"
212
+ )
213
+
214
+ import requests
215
+
216
+ payload = {
217
+ "model": self.config.model,
218
+ "messages": messages,
219
+ "stream": False,
220
+ "options": {
221
+ "temperature": temperature if temperature is not None else self.config.temperature,
222
+ "num_predict": max_tokens if max_tokens is not None else self.config.max_tokens,
223
+ }
224
+ }
225
+
226
+ try:
227
+ response = requests.post(
228
+ f"{self.config.base_url}/api/chat",
229
+ json=payload,
230
+ timeout=self.config.timeout,
231
+ )
232
+ response.raise_for_status()
233
+
234
+ return response.json().get("message", {}).get("content", "")
235
+
236
+ except requests.exceptions.RequestException as e:
237
+ raise RuntimeError(f"Ollama chat request failed: {e}")
238
+
239
+ def judge(
240
+ self,
241
+ query: str,
242
+ response: str,
243
+ criteria: str,
244
+ context: Optional[str] = None,
245
+ output_format: str = "json",
246
+ ) -> Dict[str, Any]:
247
+ """Use LLM as judge for evaluation.
248
+
249
+ Args:
250
+ query: The original query/question.
251
+ response: The response to evaluate.
252
+ criteria: The evaluation criteria/rubric.
253
+ context: Optional context/reference information.
254
+ output_format: Output format ("json" or "text").
255
+
256
+ Returns:
257
+ Dictionary with evaluation result:
258
+ - score: Float between 0 and 1
259
+ - passed: Boolean indicating if evaluation passed
260
+ - reason: Explanation of the evaluation
261
+ """
262
+ system_prompt = """You are an AI evaluation judge. Evaluate the response based on the given criteria.
263
+
264
+ Your evaluation must be fair, consistent, and based solely on the criteria provided.
265
+
266
+ Output your evaluation as JSON with the following format:
267
+ {
268
+ "score": <float between 0.0 and 1.0>,
269
+ "passed": <true or false, based on whether score >= 0.5>,
270
+ "reason": "<brief explanation of your evaluation>"
271
+ }
272
+
273
+ Only output the JSON object, nothing else."""
274
+
275
+ user_prompt = f"""## Evaluation Criteria
276
+ {criteria}
277
+
278
+ ## Query
279
+ {query}
280
+
281
+ ## Response to Evaluate
282
+ {response}
283
+ """
284
+
285
+ if context:
286
+ user_prompt += f"""
287
+ ## Context/Reference
288
+ {context}
289
+ """
290
+
291
+ user_prompt += """
292
+ ## Your Evaluation (JSON only)"""
293
+
294
+ try:
295
+ result_text = self.generate(user_prompt, system=system_prompt)
296
+ return self._parse_judge_response(result_text)
297
+
298
+ except Exception as e:
299
+ logger.error(f"LLM judge failed: {e}")
300
+ return {
301
+ "score": 0.0,
302
+ "passed": False,
303
+ "reason": f"Evaluation failed: {str(e)}",
304
+ "error": True,
305
+ }
306
+
307
+ def _parse_judge_response(self, response: str) -> Dict[str, Any]:
308
+ """Parse the LLM judge response into structured format.
309
+
310
+ Args:
311
+ response: The raw LLM response text.
312
+
313
+ Returns:
314
+ Parsed evaluation result dictionary.
315
+ """
316
+ response = response.strip()
317
+
318
+ # Try direct JSON parse first
319
+ try:
320
+ result = json.loads(response)
321
+ return self._validate_judge_result(result)
322
+ except json.JSONDecodeError:
323
+ pass
324
+
325
+ # Try to extract JSON from markdown code block
326
+ code_block_pattern = r"```(?:json)?\s*([\s\S]*?)```"
327
+ match = re.search(code_block_pattern, response)
328
+ if match:
329
+ try:
330
+ result = json.loads(match.group(1).strip())
331
+ return self._validate_judge_result(result)
332
+ except json.JSONDecodeError:
333
+ pass
334
+
335
+ # Try to find JSON object in response
336
+ json_pattern = r"\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}"
337
+ match = re.search(json_pattern, response, re.DOTALL)
338
+ if match:
339
+ try:
340
+ result = json.loads(match.group())
341
+ return self._validate_judge_result(result)
342
+ except json.JSONDecodeError:
343
+ pass
344
+
345
+ # Fallback: try to extract score and reason from text
346
+ score = 0.5
347
+ score_pattern = r"(?:score|rating)[:\s]*([0-9]*\.?[0-9]+)"
348
+ score_match = re.search(score_pattern, response, re.IGNORECASE)
349
+ if score_match:
350
+ try:
351
+ score = float(score_match.group(1))
352
+ # Normalize if score is > 1 (e.g., 1-10 scale)
353
+ if score > 1:
354
+ score = score / 10
355
+ score = max(0.0, min(1.0, score))
356
+ except ValueError:
357
+ pass
358
+
359
+ return {
360
+ "score": score,
361
+ "passed": score >= 0.5,
362
+ "reason": response[:500] if response else "Unable to parse evaluation",
363
+ "parse_error": True,
364
+ }
365
+
366
+ def _validate_judge_result(self, result: Dict[str, Any]) -> Dict[str, Any]:
367
+ """Validate and normalize the judge result.
368
+
369
+ Args:
370
+ result: Raw parsed result dictionary.
371
+
372
+ Returns:
373
+ Validated and normalized result.
374
+ """
375
+ score = result.get("score", 0.5)
376
+
377
+ # Handle various score formats
378
+ if isinstance(score, str):
379
+ try:
380
+ score = float(score)
381
+ except ValueError:
382
+ score = 0.5
383
+
384
+ # Normalize score to 0-1 range
385
+ if score > 1:
386
+ score = score / 10 if score <= 10 else score / 100
387
+ score = max(0.0, min(1.0, float(score)))
388
+
389
+ passed = result.get("passed")
390
+ if passed is None:
391
+ passed = score >= 0.5
392
+ elif isinstance(passed, str):
393
+ passed = passed.lower() in ("true", "yes", "1", "pass")
394
+
395
+ reason = result.get("reason", result.get("explanation", ""))
396
+ if not isinstance(reason, str):
397
+ reason = str(reason)
398
+
399
+ return {
400
+ "score": score,
401
+ "passed": bool(passed),
402
+ "reason": reason,
403
+ }
404
+
405
+ def batch_judge(
406
+ self,
407
+ evaluations: List[Dict[str, Any]],
408
+ ) -> List[Dict[str, Any]]:
409
+ """Run multiple judge evaluations.
410
+
411
+ Args:
412
+ evaluations: List of evaluation specifications, each containing:
413
+ - query: The query
414
+ - response: The response to evaluate
415
+ - criteria: Evaluation criteria
416
+ - context: Optional context
417
+
418
+ Returns:
419
+ List of evaluation results.
420
+ """
421
+ results = []
422
+ for eval_spec in evaluations:
423
+ result = self.judge(
424
+ query=eval_spec.get("query", ""),
425
+ response=eval_spec.get("response", ""),
426
+ criteria=eval_spec.get("criteria", ""),
427
+ context=eval_spec.get("context"),
428
+ )
429
+ results.append(result)
430
+ return results
431
+
432
+
433
+ class LocalLLMFactory:
434
+ """Factory for creating local LLM instances.
435
+
436
+ This factory provides a unified interface for creating different
437
+ types of local LLM backends.
438
+ """
439
+
440
+ _backends = {
441
+ "ollama": OllamaLLM,
442
+ }
443
+
444
+ @classmethod
445
+ def create(
446
+ cls,
447
+ backend: str = "ollama",
448
+ **kwargs,
449
+ ) -> OllamaLLM:
450
+ """Create a local LLM instance.
451
+
452
+ Args:
453
+ backend: The LLM backend to use (currently only "ollama").
454
+ **kwargs: Additional arguments passed to the LLM constructor.
455
+
456
+ Returns:
457
+ An initialized LLM instance.
458
+
459
+ Raises:
460
+ ValueError: If backend is not supported.
461
+ """
462
+ if backend not in cls._backends:
463
+ raise ValueError(
464
+ f"Unsupported LLM backend: {backend}. "
465
+ f"Supported backends: {list(cls._backends.keys())}"
466
+ )
467
+
468
+ return cls._backends[backend](**kwargs)
469
+
470
+ @classmethod
471
+ def from_string(cls, spec: str) -> OllamaLLM:
472
+ """Create a local LLM from a string specification.
473
+
474
+ Args:
475
+ spec: String specification in format "backend/model"
476
+ (e.g., "ollama/llama3.2").
477
+
478
+ Returns:
479
+ An initialized LLM instance.
480
+ """
481
+ parts = spec.split("/", 1)
482
+ backend = parts[0].lower()
483
+ model = parts[1] if len(parts) > 1 else None
484
+
485
+ config = LocalLLMConfig()
486
+ if model:
487
+ config.model = model
488
+
489
+ return cls.create(backend, config=config)
@@ -0,0 +1,19 @@
1
+ """Local metrics module.
2
+
3
+ This module provides the base class and utilities for local metric implementations.
4
+ The actual metric implementations live in fi.evals.metrics.heuristics/ and are
5
+ registered via the LocalMetricRegistry.
6
+ """
7
+
8
+ # Re-export commonly used types for convenience
9
+ from ...types import TextMetricInput, JsonMetricInput, EvalResult, BatchRunResult
10
+ from ...metrics.base_metric import BaseMetric
11
+
12
+
13
+ __all__ = [
14
+ "BaseMetric",
15
+ "TextMetricInput",
16
+ "JsonMetricInput",
17
+ "EvalResult",
18
+ "BatchRunResult",
19
+ ]