agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/cli/commands/run.py ADDED
@@ -0,0 +1,486 @@
1
+ """Run command for executing evaluations."""
2
+
3
+ import os
4
+ import sys
5
+ from pathlib import Path
6
+ from typing import Optional, Dict, Any
7
+ from enum import Enum
8
+
9
+ import typer
10
+ from rich.progress import Progress, SpinnerColumn, TextColumn
11
+
12
+ from fi.cli.config.loader import load_config, load_test_data
13
+ from fi.cli.output.formatters import format_results
14
+ from fi.cli.output.reporters import ResultReporter
15
+ from fi.cli.utils.console import console, print_error, print_success, print_warning
16
+ from fi.cli.assertions import (
17
+ AssertionEvaluator,
18
+ AssertionReporter,
19
+ ExitCode,
20
+ )
21
+
22
+
23
+ class ExecutionModeOption(str, Enum):
24
+ """Execution mode options for CLI."""
25
+ local = "local"
26
+ cloud = "cloud"
27
+ hybrid = "hybrid"
28
+
29
+
30
+ def run(
31
+ config: Optional[Path] = typer.Option(
32
+ None,
33
+ "--config", "-c",
34
+ help="Path to configuration file",
35
+ ),
36
+ eval_template: Optional[str] = typer.Option(
37
+ None,
38
+ "--eval", "-e",
39
+ help="Evaluation template to run (overrides config)",
40
+ ),
41
+ data: Optional[Path] = typer.Option(
42
+ None,
43
+ "--data", "-d",
44
+ help="Path to test data file (overrides config)",
45
+ ),
46
+ output: str = typer.Option(
47
+ "table",
48
+ "--output", "-o",
49
+ help="Output format: table, json, csv, html",
50
+ ),
51
+ parallel: int = typer.Option(
52
+ 8,
53
+ "--parallel", "-p",
54
+ help="Number of parallel workers",
55
+ ),
56
+ timeout: int = typer.Option(
57
+ 200,
58
+ "--timeout", "-T",
59
+ help="Timeout per evaluation in seconds",
60
+ ),
61
+ model: Optional[str] = typer.Option(
62
+ None,
63
+ "--model", "-m",
64
+ help="Model for LLM-as-judge evaluations",
65
+ ),
66
+ dry_run: bool = typer.Option(
67
+ False,
68
+ "--dry-run",
69
+ help="Validate config without running evaluations",
70
+ ),
71
+ output_file: Optional[Path] = typer.Option(
72
+ None,
73
+ "--output-file", "-O",
74
+ help="Path to save output file",
75
+ ),
76
+ quiet: bool = typer.Option(
77
+ False,
78
+ "--quiet", "-q",
79
+ help="Suppress progress output",
80
+ ),
81
+ no_save: bool = typer.Option(
82
+ False,
83
+ "--no-save",
84
+ help="Don't save run to history (for 'fi view' command)",
85
+ ),
86
+ check_assertions: bool = typer.Option(
87
+ True,
88
+ "--check/--no-check",
89
+ help="Check assertions after evaluation (if configured)",
90
+ ),
91
+ fail_fast: bool = typer.Option(
92
+ False,
93
+ "--fail-fast",
94
+ help="Stop on first assertion failure",
95
+ ),
96
+ strict: bool = typer.Option(
97
+ False,
98
+ "--strict",
99
+ help="Exit with error code on assertion warnings",
100
+ ),
101
+ mode: ExecutionModeOption = typer.Option(
102
+ ExecutionModeOption.cloud,
103
+ "--mode",
104
+ help="Execution mode: local (no API), cloud (API only), or hybrid (auto-route)",
105
+ ),
106
+ local_llm: Optional[str] = typer.Option(
107
+ None,
108
+ "--local-llm",
109
+ help="Local LLM for LLM-based evals (e.g., 'ollama/llama3.2')",
110
+ ),
111
+ offline: bool = typer.Option(
112
+ False,
113
+ "--offline",
114
+ help="Run in offline mode (no cloud API calls)",
115
+ ),
116
+ ) -> None:
117
+ """
118
+ Run evaluations from config file or CLI arguments.
119
+
120
+ Examples:
121
+ fi run # Use default config (cloud mode)
122
+ fi run -c custom.yaml # Use custom config
123
+ fi run -e groundedness -d data.json # Run single evaluation
124
+ fi run -o json > results.json # Output as JSON
125
+ fi run --mode local # Run only local heuristic metrics
126
+ fi run --mode hybrid # Auto-route between local and cloud
127
+ fi run --local-llm ollama/llama3.2 # Use local LLM for LLM-based evals
128
+ fi run --offline # No cloud API calls (implies local mode)
129
+ """
130
+ from fi.evals.evaluator import Evaluator
131
+ from fi.evals.local import HybridEvaluator
132
+
133
+ # Handle offline mode implications
134
+ effective_mode = mode
135
+ if offline:
136
+ if mode == ExecutionModeOption.cloud:
137
+ effective_mode = ExecutionModeOption.local
138
+ print_warning("Offline mode enabled - switching from cloud to local mode")
139
+
140
+ # Check for API keys (only required for cloud/hybrid modes)
141
+ api_key = os.environ.get("FI_API_KEY")
142
+ secret_key = os.environ.get("FI_SECRET_KEY")
143
+
144
+ if effective_mode != ExecutionModeOption.local and (not api_key or not secret_key):
145
+ if effective_mode == ExecutionModeOption.cloud:
146
+ print_warning(
147
+ "API keys not found in environment.\n"
148
+ "Set FI_API_KEY and FI_SECRET_KEY environment variables."
149
+ )
150
+ if not dry_run:
151
+ raise typer.Exit(1)
152
+ elif effective_mode == ExecutionModeOption.hybrid:
153
+ print_warning(
154
+ "API keys not found - hybrid mode will only run local metrics."
155
+ )
156
+
157
+ # Load configuration or use CLI arguments
158
+ if eval_template and data:
159
+ # CLI-only mode
160
+ test_data = load_test_data(data)
161
+ evaluations = [{"template": eval_template, "data": test_data}]
162
+ defaults = {"timeout": timeout, "parallel_workers": parallel}
163
+ else:
164
+ # Config file mode
165
+ try:
166
+ eval_config = load_config(config)
167
+ except FileNotFoundError as e:
168
+ print_error(str(e))
169
+ raise typer.Exit(1)
170
+ except ValueError as e:
171
+ print_error(f"Configuration error: {e}")
172
+ raise typer.Exit(1)
173
+
174
+ defaults = {
175
+ "timeout": eval_config.get_defaults().timeout,
176
+ "parallel_workers": eval_config.get_defaults().parallel_workers,
177
+ "model": eval_config.get_defaults().model,
178
+ }
179
+
180
+ # Override with CLI arguments
181
+ if timeout != 200:
182
+ defaults["timeout"] = timeout
183
+ if parallel != 8:
184
+ defaults["parallel_workers"] = parallel
185
+ if model:
186
+ defaults["model"] = model
187
+
188
+ # Build evaluations list
189
+ evaluations = []
190
+ for eval_def in eval_config.evaluations:
191
+ try:
192
+ test_data = load_test_data(eval_def.data)
193
+ except FileNotFoundError:
194
+ print_error(f"Data file not found: {eval_def.data}")
195
+ raise typer.Exit(1)
196
+
197
+ templates = eval_def.templates or [eval_def.template]
198
+ for template in templates:
199
+ if template:
200
+ evaluations.append({
201
+ "name": eval_def.name,
202
+ "template": template,
203
+ "data": test_data,
204
+ "config": eval_def.config,
205
+ })
206
+
207
+ if dry_run:
208
+ print_success(
209
+ f"Configuration valid!\n"
210
+ f"Would run {len(evaluations)} evaluation(s) in {effective_mode.value} mode."
211
+ )
212
+ return
213
+
214
+ # Initialize local LLM if specified
215
+ llm_instance = None
216
+ if local_llm:
217
+ try:
218
+ from fi.evals.local import LocalLLMFactory
219
+ llm_instance = LocalLLMFactory.from_string(local_llm)
220
+ if not llm_instance.is_available():
221
+ print_warning(
222
+ f"Local LLM '{local_llm}' is not available. "
223
+ "Make sure Ollama is running: `ollama serve`"
224
+ )
225
+ llm_instance = None
226
+ elif not quiet:
227
+ print_success(f"Local LLM initialized: {local_llm}")
228
+ except Exception as e:
229
+ print_warning(f"Failed to initialize local LLM: {e}")
230
+ llm_instance = None
231
+
232
+ # Initialize evaluator(s) based on mode
233
+ cloud_evaluator = None
234
+ hybrid_evaluator = None
235
+
236
+ if effective_mode == ExecutionModeOption.cloud:
237
+ # Pure cloud mode - use standard evaluator
238
+ cloud_evaluator = Evaluator(
239
+ fi_api_key=api_key,
240
+ fi_secret_key=secret_key,
241
+ max_workers=defaults["parallel_workers"],
242
+ )
243
+ elif effective_mode == ExecutionModeOption.local:
244
+ # Pure local mode - use hybrid evaluator in offline mode
245
+ hybrid_evaluator = HybridEvaluator(
246
+ local_llm=llm_instance,
247
+ prefer_local=True,
248
+ fallback_to_cloud=False,
249
+ offline_mode=True,
250
+ )
251
+ if not quiet:
252
+ print_success("Running in local mode (no cloud API calls)")
253
+ else:
254
+ # Hybrid mode - use hybrid evaluator with cloud fallback
255
+ if api_key and secret_key:
256
+ cloud_evaluator = Evaluator(
257
+ fi_api_key=api_key,
258
+ fi_secret_key=secret_key,
259
+ max_workers=defaults["parallel_workers"],
260
+ )
261
+ hybrid_evaluator = HybridEvaluator(
262
+ local_llm=llm_instance,
263
+ cloud_evaluator=cloud_evaluator,
264
+ prefer_local=True,
265
+ fallback_to_cloud=not offline,
266
+ offline_mode=offline,
267
+ )
268
+ if not quiet:
269
+ print_success("Running in hybrid mode (auto-routing local/cloud)")
270
+
271
+ all_results = []
272
+ local_metrics_run = 0
273
+ cloud_metrics_run = 0
274
+
275
+ # Helper function to run single evaluation
276
+ def run_evaluation(eval_def: Dict[str, Any]) -> None:
277
+ nonlocal local_metrics_run, cloud_metrics_run
278
+
279
+ template = eval_def["template"]
280
+ data = eval_def["data"]
281
+
282
+ if effective_mode == ExecutionModeOption.cloud:
283
+ # Pure cloud mode
284
+ results = cloud_evaluator.evaluate(
285
+ eval_templates=template,
286
+ inputs=data,
287
+ timeout=defaults["timeout"],
288
+ model_name=defaults.get("model"),
289
+ )
290
+ all_results.extend(results.eval_results)
291
+ cloud_metrics_run += 1
292
+
293
+ elif effective_mode == ExecutionModeOption.local:
294
+ # Pure local mode
295
+ result = hybrid_evaluator.evaluate(
296
+ template=template,
297
+ inputs=data,
298
+ config=eval_def.get("config", {}),
299
+ )
300
+ all_results.extend(result.results.eval_results)
301
+ local_metrics_run += len(result.executed_locally)
302
+
303
+ else:
304
+ # Hybrid mode - route based on metric type
305
+ from fi.evals.local import can_run_locally
306
+
307
+ if can_run_locally(template) or hybrid_evaluator.can_use_local_llm(template):
308
+ # Run locally
309
+ result = hybrid_evaluator.evaluate(
310
+ template=template,
311
+ inputs=data,
312
+ config=eval_def.get("config", {}),
313
+ )
314
+ all_results.extend(result.results.eval_results)
315
+ local_metrics_run += len(result.executed_locally)
316
+ elif cloud_evaluator:
317
+ # Run in cloud
318
+ results = cloud_evaluator.evaluate(
319
+ eval_templates=template,
320
+ inputs=data,
321
+ timeout=defaults["timeout"],
322
+ model_name=defaults.get("model"),
323
+ )
324
+ all_results.extend(results.eval_results)
325
+ cloud_metrics_run += 1
326
+ else:
327
+ # No cloud available
328
+ from fi.evals.types import EvalResult
329
+ for _ in data:
330
+ all_results.append(
331
+ EvalResult(
332
+ name=template,
333
+ output=None,
334
+ reason="Cloud unavailable - cannot run this metric locally",
335
+ runtime=0,
336
+ )
337
+ )
338
+
339
+ # Run evaluations
340
+ if not quiet:
341
+ with Progress(
342
+ SpinnerColumn(),
343
+ TextColumn("[progress.description]{task.description}"),
344
+ console=console,
345
+ ) as progress:
346
+ for eval_def in evaluations:
347
+ task = progress.add_task(
348
+ f"Running {eval_def['template']}...",
349
+ total=None
350
+ )
351
+
352
+ try:
353
+ run_evaluation(eval_def)
354
+ progress.update(task, description=f"[green]✓[/green] {eval_def['template']}")
355
+ except Exception as e:
356
+ progress.update(task, description=f"[red]✗[/red] {eval_def['template']}: {e}")
357
+ if not quiet:
358
+ console.print(f"[red]Error running {eval_def['template']}: {e}[/red]")
359
+
360
+ progress.remove_task(task)
361
+ else:
362
+ for eval_def in evaluations:
363
+ try:
364
+ run_evaluation(eval_def)
365
+ except Exception as e:
366
+ console.print(f"[red]Error: {e}[/red]", file=sys.stderr)
367
+
368
+ # Show execution summary in hybrid mode
369
+ if not quiet and effective_mode == ExecutionModeOption.hybrid:
370
+ console.print(
371
+ f"\n[dim]Executed: {local_metrics_run} local, {cloud_metrics_run} cloud[/dim]"
372
+ )
373
+
374
+ # Create combined results
375
+ from fi.evals.types import BatchRunResult
376
+ combined_results = BatchRunResult(eval_results=all_results)
377
+
378
+ # Format and display results
379
+ output_path = str(output_file) if output_file else None
380
+ result_str = format_results(combined_results, output, console, output_path)
381
+
382
+ # For non-table formats, print the result
383
+ if output != "table" and result_str:
384
+ if output_file:
385
+ print_success(f"Results saved to: {output_file}")
386
+ else:
387
+ console.print(result_str)
388
+
389
+ # Print summary for table output
390
+ if output == "table" and not quiet:
391
+ reporter = ResultReporter(console)
392
+ reporter.report_summary(combined_results)
393
+
394
+ # Save run to history
395
+ if not no_save and all_results:
396
+ from fi.cli.storage import RunHistory
397
+
398
+ history = RunHistory()
399
+ templates_used = list(set(e["template"] for e in evaluations))
400
+ config_path_str = str(config) if config else None
401
+
402
+ record = history.save_run(
403
+ results=combined_results,
404
+ config_file=config_path_str,
405
+ templates=templates_used,
406
+ )
407
+
408
+ if not quiet:
409
+ console.print(f"\n[dim]Run saved: {record.run_id}[/dim]")
410
+ console.print("[dim]View with: fi view --last[/dim]")
411
+
412
+ # Check assertions if configured
413
+ if check_assertions and 'eval_config' in dir() and eval_config is not None:
414
+ assertion_config = _build_assertion_config(eval_config, fail_fast)
415
+
416
+ if assertion_config.get('assertions') or assertion_config.get('thresholds', {}).get('default_pass_rate'):
417
+ # Convert results to dict format for evaluator
418
+ results_dict = {
419
+ "eval_results": [
420
+ {
421
+ "name": r.name,
422
+ "output": r.output,
423
+ "reason": r.reason,
424
+ "runtime": r.runtime,
425
+ "output_type": r.output_type,
426
+ "eval_id": r.eval_id,
427
+ }
428
+ for r in all_results
429
+ ]
430
+ }
431
+
432
+ evaluator = AssertionEvaluator(results_dict, assertion_config)
433
+ report = evaluator.evaluate_all()
434
+
435
+ # Display assertion report
436
+ if not quiet and report.total_assertions > 0:
437
+ reporter = AssertionReporter(console)
438
+ console.print() # Blank line before assertions
439
+ reporter.display(report)
440
+ reporter.display_summary_line(report)
441
+
442
+ # Determine exit code based on assertion results
443
+ if report.failed > 0:
444
+ raise typer.Exit(ExitCode.ASSERTION_FAILED)
445
+ elif report.warnings > 0 and strict:
446
+ raise typer.Exit(ExitCode.ASSERTION_WARNING)
447
+
448
+
449
+ def _build_assertion_config(
450
+ eval_config,
451
+ fail_fast: bool = False
452
+ ) -> Dict[str, Any]:
453
+ """Build assertion config dictionary from FIEvaluationConfig.
454
+
455
+ Args:
456
+ eval_config: The loaded FIEvaluationConfig object.
457
+ fail_fast: Whether to enable fail-fast mode.
458
+
459
+ Returns:
460
+ Dictionary with 'assertions' and 'thresholds' keys.
461
+ """
462
+ assertion_config: Dict[str, Any] = {
463
+ "assertions": [],
464
+ "thresholds": {}
465
+ }
466
+
467
+ # Convert assertion configs
468
+ if eval_config.assertions:
469
+ for assertion in eval_config.assertions:
470
+ assertion_dict = {
471
+ "template": assertion.template,
472
+ "global": assertion.is_global,
473
+ "conditions": assertion.conditions,
474
+ "on_fail": assertion.on_fail,
475
+ }
476
+ assertion_config["assertions"].append(assertion_dict)
477
+
478
+ # Convert thresholds config
479
+ if eval_config.thresholds:
480
+ assertion_config["thresholds"] = {
481
+ "default_pass_rate": eval_config.thresholds.default_pass_rate,
482
+ "fail_fast": fail_fast or eval_config.thresholds.fail_fast,
483
+ "overrides": eval_config.thresholds.overrides or {},
484
+ }
485
+
486
+ return assertion_config
@@ -0,0 +1,173 @@
1
+ """Validate command for checking configuration files."""
2
+
3
+ import os
4
+ from pathlib import Path
5
+ from typing import Optional
6
+
7
+ import typer
8
+
9
+ from fi.cli.config.loader import load_config, load_test_data, find_config_file
10
+ from fi.cli.utils.console import console, print_error, print_success, print_warning
11
+
12
+
13
+ def validate(
14
+ config: Optional[Path] = typer.Option(
15
+ None,
16
+ "--config", "-c",
17
+ help="Path to configuration file",
18
+ ),
19
+ strict: bool = typer.Option(
20
+ False,
21
+ "--strict", "-s",
22
+ help="Enable strict validation mode",
23
+ ),
24
+ ) -> None:
25
+ """
26
+ Validate configuration file and test data.
27
+
28
+ Checks:
29
+ - YAML syntax validity
30
+ - Template name existence
31
+ - Required input fields for templates
32
+ - Data file accessibility
33
+ - API key presence (warning if missing)
34
+ """
35
+ errors = []
36
+ warnings = []
37
+
38
+ # Find config file
39
+ if config:
40
+ config_path = Path(config)
41
+ else:
42
+ config_path = find_config_file()
43
+
44
+ if not config_path:
45
+ print_error("No configuration file found.")
46
+ raise typer.Exit(1)
47
+
48
+ console.print(f"[dim]Validating: {config_path}[/dim]\n")
49
+
50
+ # 1. Load and validate config
51
+ try:
52
+ eval_config = load_config(config_path)
53
+ console.print("[green]✓[/green] Configuration file syntax is valid")
54
+ except FileNotFoundError as e:
55
+ errors.append(f"Configuration file not found: {e}")
56
+ except ValueError as e:
57
+ errors.append(f"Configuration validation failed: {e}")
58
+
59
+ if errors:
60
+ _print_validation_results(errors, warnings, strict)
61
+ raise typer.Exit(1)
62
+
63
+ # 2. Validate templates exist
64
+ from fi.evals import templates as templates_module
65
+ from fi.evals.templates import EvalTemplate
66
+
67
+ available_templates = set()
68
+ for name in dir(templates_module):
69
+ obj = getattr(templates_module, name)
70
+ if (
71
+ isinstance(obj, type)
72
+ and issubclass(obj, EvalTemplate)
73
+ and obj is not EvalTemplate
74
+ and hasattr(obj, "eval_name")
75
+ ):
76
+ available_templates.add(obj.eval_name)
77
+
78
+ for eval_def in eval_config.evaluations:
79
+ templates = eval_def.templates or ([eval_def.template] if eval_def.template else [])
80
+ for template in templates:
81
+ if template and template not in available_templates:
82
+ errors.append(
83
+ f"Unknown template '{template}' in evaluation '{eval_def.name}'. "
84
+ f"Run 'fi list templates' to see available templates."
85
+ )
86
+
87
+ if not errors:
88
+ console.print("[green]✓[/green] All template names are valid")
89
+
90
+ # 3. Validate data files
91
+ data_files_valid = True
92
+ for eval_def in eval_config.evaluations:
93
+ data_path = Path(eval_def.data)
94
+
95
+ # Check if path is relative to config file
96
+ if not data_path.is_absolute():
97
+ data_path = config_path.parent / data_path
98
+
99
+ if not data_path.exists():
100
+ errors.append(f"Data file not found: {eval_def.data}")
101
+ data_files_valid = False
102
+ else:
103
+ try:
104
+ test_data = load_test_data(data_path)
105
+ if not test_data:
106
+ warnings.append(f"Data file is empty: {eval_def.data}")
107
+ elif len(test_data) == 0:
108
+ warnings.append(f"No test cases in: {eval_def.data}")
109
+ except Exception as e:
110
+ errors.append(f"Error loading data file {eval_def.data}: {e}")
111
+ data_files_valid = False
112
+
113
+ if data_files_valid and not errors:
114
+ console.print("[green]✓[/green] All data files are accessible")
115
+
116
+ # 4. Check API keys
117
+ api_key = os.environ.get("FI_API_KEY")
118
+ secret_key = os.environ.get("FI_SECRET_KEY")
119
+
120
+ if not api_key:
121
+ warnings.append("FI_API_KEY environment variable not set")
122
+ if not secret_key:
123
+ warnings.append("FI_SECRET_KEY environment variable not set")
124
+
125
+ if api_key and secret_key:
126
+ console.print("[green]✓[/green] API keys are configured")
127
+ else:
128
+ console.print("[yellow]![/yellow] API keys not configured (evaluations will fail)")
129
+
130
+ # 5. Validate output configuration
131
+ if eval_config.output:
132
+ output_path = Path(eval_config.output.path)
133
+ if not output_path.is_absolute():
134
+ output_path = config_path.parent / output_path
135
+
136
+ if output_path.exists() and not output_path.is_dir():
137
+ warnings.append(f"Output path exists but is not a directory: {eval_config.output.path}")
138
+ elif not output_path.exists():
139
+ warnings.append(f"Output directory does not exist (will be created): {eval_config.output.path}")
140
+
141
+ # Print results
142
+ _print_validation_results(errors, warnings, strict)
143
+
144
+ if errors:
145
+ raise typer.Exit(1)
146
+ elif strict and warnings:
147
+ raise typer.Exit(1)
148
+
149
+
150
+ def _print_validation_results(errors: list, warnings: list, strict: bool) -> None:
151
+ """Print validation errors and warnings."""
152
+ console.print()
153
+
154
+ if errors:
155
+ console.print("[bold red]Errors:[/bold red]")
156
+ for error in errors:
157
+ console.print(f" [red]✗[/red] {error}")
158
+
159
+ if warnings:
160
+ console.print("[bold yellow]Warnings:[/bold yellow]")
161
+ for warning in warnings:
162
+ console.print(f" [yellow]![/yellow] {warning}")
163
+
164
+ console.print()
165
+
166
+ if errors:
167
+ print_error(f"Validation failed with {len(errors)} error(s)")
168
+ elif warnings and strict:
169
+ print_warning(f"Validation failed with {len(warnings)} warning(s) (strict mode)")
170
+ elif warnings:
171
+ print_warning(f"Validation passed with {len(warnings)} warning(s)")
172
+ else:
173
+ print_success("Validation passed!")