agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,155 @@
1
+ """Configuration file loading and discovery."""
2
+
3
+ from pathlib import Path
4
+ from typing import Optional, Union
5
+
6
+ import yaml
7
+ from pydantic import ValidationError
8
+
9
+ from fi.cli.config.schema import FIEvaluationConfig
10
+
11
+
12
+ # Default config file names to look for
13
+ CONFIG_FILE_NAMES = [
14
+ "fi-evaluation.yaml",
15
+ "fi-evaluation.yml",
16
+ ".fi-evaluation.yaml",
17
+ ".fi-evaluation.yml",
18
+ ]
19
+
20
+
21
+ def find_config_file(start_path: Optional[Path] = None) -> Optional[Path]:
22
+ """
23
+ Find a configuration file by searching current directory and parents.
24
+
25
+ Args:
26
+ start_path: Starting directory for search (default: current working directory)
27
+
28
+ Returns:
29
+ Path to configuration file if found, None otherwise
30
+ """
31
+ if start_path is None:
32
+ start_path = Path.cwd()
33
+
34
+ current = start_path.resolve()
35
+
36
+ # Search up the directory tree
37
+ while current != current.parent:
38
+ for config_name in CONFIG_FILE_NAMES:
39
+ config_path = current / config_name
40
+ if config_path.exists():
41
+ return config_path
42
+ current = current.parent
43
+
44
+ # Check root as well
45
+ for config_name in CONFIG_FILE_NAMES:
46
+ config_path = current / config_name
47
+ if config_path.exists():
48
+ return config_path
49
+
50
+ return None
51
+
52
+
53
+ def load_config(
54
+ config_path: Optional[Union[str, Path]] = None
55
+ ) -> FIEvaluationConfig:
56
+ """
57
+ Load and validate configuration from a YAML file.
58
+
59
+ Args:
60
+ config_path: Path to configuration file. If not provided,
61
+ will search for config file automatically.
62
+
63
+ Returns:
64
+ Validated FIEvaluationConfig object
65
+
66
+ Raises:
67
+ FileNotFoundError: If config file not found
68
+ ValueError: If config file is invalid YAML
69
+ ValidationError: If config doesn't match schema
70
+ """
71
+ if config_path is None:
72
+ config_path = find_config_file()
73
+ if config_path is None:
74
+ raise FileNotFoundError(
75
+ "No configuration file found. "
76
+ "Create a fi-evaluation.yaml file or specify --config path."
77
+ )
78
+ else:
79
+ config_path = Path(config_path)
80
+
81
+ if not config_path.exists():
82
+ raise FileNotFoundError(f"Configuration file not found: {config_path}")
83
+
84
+ # Load YAML
85
+ try:
86
+ with open(config_path, "r") as f:
87
+ raw_config = yaml.safe_load(f)
88
+ except yaml.YAMLError as e:
89
+ raise ValueError(f"Invalid YAML in configuration file: {e}")
90
+
91
+ if raw_config is None:
92
+ raise ValueError("Configuration file is empty")
93
+
94
+ # Validate against schema
95
+ try:
96
+ config = FIEvaluationConfig(**raw_config)
97
+ except ValidationError as e:
98
+ raise ValueError(f"Configuration validation failed: {e}")
99
+
100
+ return config
101
+
102
+
103
+ def load_test_data(data_path: Union[str, Path]) -> list:
104
+ """
105
+ Load test data from a JSON, JSONL, or CSV file.
106
+
107
+ Args:
108
+ data_path: Path to test data file
109
+
110
+ Returns:
111
+ List of test case dictionaries
112
+
113
+ Raises:
114
+ FileNotFoundError: If data file not found
115
+ ValueError: If data format is unsupported or invalid
116
+ """
117
+ import json
118
+ import csv
119
+
120
+ data_path = Path(data_path)
121
+
122
+ if not data_path.exists():
123
+ raise FileNotFoundError(f"Test data file not found: {data_path}")
124
+
125
+ suffix = data_path.suffix.lower()
126
+
127
+ if suffix == ".json":
128
+ with open(data_path, "r") as f:
129
+ data = json.load(f)
130
+ if isinstance(data, dict):
131
+ return [data]
132
+ return data
133
+
134
+ elif suffix == ".jsonl":
135
+ data = []
136
+ with open(data_path, "r") as f:
137
+ for line in f:
138
+ line = line.strip()
139
+ if line:
140
+ data.append(json.loads(line))
141
+ return data
142
+
143
+ elif suffix == ".csv":
144
+ data = []
145
+ with open(data_path, "r", newline="") as f:
146
+ reader = csv.DictReader(f)
147
+ for row in reader:
148
+ data.append(dict(row))
149
+ return data
150
+
151
+ else:
152
+ raise ValueError(
153
+ f"Unsupported data format: {suffix}. "
154
+ "Supported formats: .json, .jsonl, .csv"
155
+ )
@@ -0,0 +1,174 @@
1
+ """Configuration schema definitions for fi-evaluation.yaml."""
2
+
3
+ from typing import List, Optional, Dict, Any
4
+ from pydantic import BaseModel, ConfigDict, Field, field_validator
5
+
6
+
7
+ class APIConfig(BaseModel):
8
+ """API configuration settings."""
9
+ base_url: str = Field(
10
+ default="https://api.futureagi.com",
11
+ description="Base URL for the Future AGI API"
12
+ )
13
+
14
+
15
+ class DefaultsConfig(BaseModel):
16
+ """Default settings for evaluations."""
17
+ model: str = Field(
18
+ default="gpt-4o",
19
+ description="Default model for LLM-as-judge evaluations"
20
+ )
21
+ timeout: int = Field(
22
+ default=200,
23
+ description="Default timeout in seconds"
24
+ )
25
+ parallel_workers: int = Field(
26
+ default=8,
27
+ description="Number of parallel workers for evaluation"
28
+ )
29
+
30
+
31
+ class EvaluationConfig(BaseModel):
32
+ """Configuration for a single evaluation."""
33
+ name: str = Field(..., description="Name of this evaluation")
34
+ template: Optional[str] = Field(
35
+ default=None,
36
+ description="Single evaluation template to use"
37
+ )
38
+ templates: Optional[List[str]] = Field(
39
+ default=None,
40
+ description="List of evaluation templates to use"
41
+ )
42
+ data: str = Field(..., description="Path to test data file")
43
+ config: Optional[Dict[str, Any]] = Field(
44
+ default=None,
45
+ description="Additional configuration for the evaluation"
46
+ )
47
+
48
+ @field_validator("templates", "template")
49
+ @classmethod
50
+ def validate_template_presence(cls, v, info):
51
+ """Ensure at least one template specification method is used."""
52
+ return v
53
+
54
+
55
+ class AssertionConfig(BaseModel):
56
+ """Configuration for evaluation assertions."""
57
+ template: Optional[str] = Field(
58
+ default=None,
59
+ description="Template to assert on (mutually exclusive with 'global')"
60
+ )
61
+ conditions: List[str] = Field(
62
+ default_factory=list,
63
+ description="List of assertion conditions (e.g., 'pass_rate >= 0.85')"
64
+ )
65
+ on_fail: str = Field(
66
+ default="error",
67
+ description="Action on failure: error, warn, or skip"
68
+ )
69
+ is_global: bool = Field(
70
+ default=False,
71
+ alias="global",
72
+ description="If true, assertion applies globally across all templates"
73
+ )
74
+
75
+ model_config = ConfigDict(populate_by_name=True)
76
+
77
+ @field_validator("on_fail")
78
+ @classmethod
79
+ def validate_on_fail(cls, v):
80
+ """Validate on_fail value."""
81
+ valid_values = ["warn", "error", "skip"]
82
+ if v not in valid_values:
83
+ raise ValueError(f"on_fail must be one of: {valid_values}")
84
+ return v
85
+
86
+
87
+ class ThresholdOverrides(BaseModel):
88
+ """Per-template threshold overrides."""
89
+ model_config = ConfigDict(extra="allow")
90
+
91
+
92
+ class ThresholdsConfig(BaseModel):
93
+ """Threshold shortcuts for assertions."""
94
+ default_pass_rate: Optional[float] = Field(
95
+ default=None,
96
+ description="Default pass rate threshold for all templates"
97
+ )
98
+ fail_fast: bool = Field(
99
+ default=False,
100
+ description="Stop on first assertion failure"
101
+ )
102
+ overrides: Optional[Dict[str, float]] = Field(
103
+ default=None,
104
+ description="Per-template threshold overrides"
105
+ )
106
+
107
+
108
+ class OutputConfig(BaseModel):
109
+ """Output configuration settings."""
110
+ format: str = Field(
111
+ default="json",
112
+ description="Output format: json, table, csv, html"
113
+ )
114
+ path: str = Field(
115
+ default="./results/",
116
+ description="Path to save results"
117
+ )
118
+ include_metadata: bool = Field(
119
+ default=True,
120
+ description="Include metadata in output"
121
+ )
122
+
123
+ @field_validator("format")
124
+ @classmethod
125
+ def validate_format(cls, v):
126
+ """Validate output format."""
127
+ valid_formats = ["json", "table", "csv", "html"]
128
+ if v not in valid_formats:
129
+ raise ValueError(f"format must be one of: {valid_formats}")
130
+ return v
131
+
132
+
133
+ class FIEvaluationConfig(BaseModel):
134
+ """Root configuration schema for fi-evaluation.yaml."""
135
+ version: str = Field(
136
+ default="1.0",
137
+ description="Configuration file version"
138
+ )
139
+ api: Optional[APIConfig] = Field(
140
+ default=None,
141
+ description="API configuration"
142
+ )
143
+ defaults: Optional[DefaultsConfig] = Field(
144
+ default=None,
145
+ description="Default evaluation settings"
146
+ )
147
+ evaluations: List[EvaluationConfig] = Field(
148
+ ...,
149
+ description="List of evaluation configurations"
150
+ )
151
+ output: Optional[OutputConfig] = Field(
152
+ default=None,
153
+ description="Output configuration"
154
+ )
155
+ assertions: Optional[List[AssertionConfig]] = Field(
156
+ default=None,
157
+ description="Assertions to run on evaluation results"
158
+ )
159
+ thresholds: Optional[ThresholdsConfig] = Field(
160
+ default=None,
161
+ description="Threshold shortcuts for assertions"
162
+ )
163
+
164
+ def get_defaults(self) -> DefaultsConfig:
165
+ """Get defaults config, creating one if not present."""
166
+ return self.defaults or DefaultsConfig()
167
+
168
+ def get_output_config(self) -> OutputConfig:
169
+ """Get output config, creating one if not present."""
170
+ return self.output or OutputConfig()
171
+
172
+ def get_api_config(self) -> APIConfig:
173
+ """Get API config, creating one if not present."""
174
+ return self.api or APIConfig()
fi/cli/main.py ADDED
@@ -0,0 +1,78 @@
1
+ """
2
+ AI Evaluation CLI
3
+
4
+ Command-line interface for running LLM evaluations with 60+ templates.
5
+ """
6
+
7
+ import typer
8
+ from rich.console import Console
9
+
10
+ from fi.cli.commands.init import init_project
11
+ from fi.cli.commands.run import run
12
+ from fi.cli.commands.list_cmd import list_resources
13
+ from fi.cli.commands.validate import validate
14
+ from fi.cli.commands.config import config_app
15
+ from fi.cli.commands.view import view
16
+ from fi.cli.commands.export import export
17
+
18
+ # Create main app
19
+ app = typer.Typer(
20
+ name="fi",
21
+ help="AI Evaluation CLI - Evaluate LLM outputs with 60+ templates",
22
+ no_args_is_help=True,
23
+ rich_markup_mode="rich",
24
+ )
25
+
26
+ console = Console()
27
+
28
+
29
+ # Register commands
30
+ app.command("init", help="Initialize a new evaluation project")(init_project)
31
+ app.command("run", help="Run evaluations from config or CLI")(run)
32
+ app.command("list", help="List available templates and resources")(list_resources)
33
+ app.command("validate", help="Validate configuration file")(validate)
34
+ app.command("view", help="View evaluation results from previous runs")(view)
35
+ app.command("export", help="Export evaluation results to file")(export)
36
+ app.add_typer(config_app, name="config", help="Manage CLI configuration")
37
+
38
+
39
+ @app.callback(invoke_without_command=True)
40
+ def main_callback(
41
+ ctx: typer.Context,
42
+ version: bool = typer.Option(
43
+ False,
44
+ "--version", "-v",
45
+ help="Show version and exit",
46
+ ),
47
+ ) -> None:
48
+ """
49
+ AI Evaluation CLI by Future AGI.
50
+
51
+ Evaluate LLM outputs with 60+ pre-built evaluation templates.
52
+
53
+ Quick Start:
54
+ fi init my-project # Initialize project
55
+ fi run # Run evaluations
56
+ fi list templates # List available templates
57
+ """
58
+ if version:
59
+ from importlib.metadata import version as get_version
60
+ try:
61
+ v = get_version("ai-evaluation")
62
+ except Exception:
63
+ v = "1.0.0"
64
+ console.print(f"ai-evaluation version {v}")
65
+ raise typer.Exit()
66
+
67
+ # If no command provided and not asking for version, show help
68
+ if ctx.invoked_subcommand is None and not version:
69
+ console.print(ctx.get_help())
70
+
71
+
72
+ def main() -> None:
73
+ """Main entry point for the CLI."""
74
+ app()
75
+
76
+
77
+ if __name__ == "__main__":
78
+ main()
@@ -0,0 +1,6 @@
1
+ """Output formatters and reporters for CLI results."""
2
+
3
+ from fi.cli.output.formatters import format_results
4
+ from fi.cli.output.reporters import ResultReporter
5
+
6
+ __all__ = ["format_results", "ResultReporter"]
@@ -0,0 +1,106 @@
1
+ """Format evaluation results for CLI output."""
2
+
3
+ import csv
4
+ import io
5
+ import json
6
+ from typing import Optional
7
+
8
+ from rich.console import Console
9
+ from rich.table import Table
10
+
11
+
12
+ def format_results(
13
+ batch_result,
14
+ fmt: str,
15
+ console: Console,
16
+ output_path: Optional[str] = None,
17
+ ) -> Optional[str]:
18
+ """Format BatchRunResult for display or file output.
19
+
20
+ Args:
21
+ batch_result: BatchRunResult with eval_results list.
22
+ fmt: Output format — "table", "json", "csv", or "html".
23
+ console: Rich Console for table rendering.
24
+ output_path: Optional file path to write output to.
25
+
26
+ Returns:
27
+ Formatted string for non-table formats, None for table format.
28
+ """
29
+ results = [r for r in batch_result.eval_results if r is not None]
30
+
31
+ if fmt == "table":
32
+ _print_table(results, console)
33
+ return None
34
+ elif fmt == "json":
35
+ text = _to_json(results)
36
+ elif fmt == "csv":
37
+ text = _to_csv(results)
38
+ elif fmt == "html":
39
+ text = _to_html(results)
40
+ else:
41
+ text = _to_json(results)
42
+
43
+ if output_path:
44
+ with open(output_path, "w") as f:
45
+ f.write(text)
46
+
47
+ return text
48
+
49
+
50
+ def _print_table(results, console: Console) -> None:
51
+ table = Table(title="Evaluation Results", show_lines=True)
52
+ table.add_column("Metric", style="cyan", no_wrap=True)
53
+ table.add_column("Output", style="green")
54
+ table.add_column("Reason", style="dim")
55
+ table.add_column("Runtime (ms)", justify="right")
56
+
57
+ for r in results:
58
+ table.add_row(
59
+ r.name,
60
+ str(r.output) if r.output is not None else "-",
61
+ (r.reason or "-")[:80],
62
+ str(r.runtime),
63
+ )
64
+
65
+ console.print(table)
66
+
67
+
68
+ def _to_json(results) -> str:
69
+ rows = []
70
+ for r in results:
71
+ rows.append({
72
+ "name": r.name,
73
+ "output": r.output,
74
+ "reason": r.reason,
75
+ "runtime": r.runtime,
76
+ })
77
+ return json.dumps(rows, indent=2, default=str)
78
+
79
+
80
+ def _to_csv(results) -> str:
81
+ buf = io.StringIO()
82
+ writer = csv.writer(buf)
83
+ writer.writerow(["name", "output", "reason", "runtime"])
84
+ for r in results:
85
+ writer.writerow([r.name, r.output, r.reason, r.runtime])
86
+ return buf.getvalue()
87
+
88
+
89
+ def _to_html(results) -> str:
90
+ rows_html = ""
91
+ for r in results:
92
+ rows_html += (
93
+ f"<tr><td>{r.name}</td><td>{r.output}</td>"
94
+ f"<td>{r.reason or ''}</td><td>{r.runtime}</td></tr>\n"
95
+ )
96
+ return f"""<!DOCTYPE html>
97
+ <html><head><title>Evaluation Results</title>
98
+ <style>
99
+ body {{ font-family: sans-serif; margin: 2rem; }}
100
+ table {{ border-collapse: collapse; width: 100%; }}
101
+ th, td {{ border: 1px solid #ddd; padding: 8px; text-align: left; }}
102
+ th {{ background: #f5f5f5; }}
103
+ </style></head><body>
104
+ <h1>Evaluation Results</h1>
105
+ <table><tr><th>Metric</th><th>Output</th><th>Reason</th><th>Runtime (ms)</th></tr>
106
+ {rows_html}</table></body></html>"""
@@ -0,0 +1,46 @@
1
+ """Summary reporters for CLI evaluation results."""
2
+
3
+ from rich.console import Console
4
+ from rich.panel import Panel
5
+ from rich.table import Table
6
+
7
+
8
+ class ResultReporter:
9
+ """Print a summary panel after evaluation results are displayed."""
10
+
11
+ def __init__(self, console: Console):
12
+ self.console = console
13
+
14
+ def report_summary(self, batch_result) -> None:
15
+ """Print a summary of the batch evaluation results."""
16
+ results = [r for r in batch_result.eval_results if r is not None]
17
+ total = len(results)
18
+
19
+ if total == 0:
20
+ self.console.print("[yellow]No results to summarise.[/yellow]")
21
+ return
22
+
23
+ # Count numeric scores (output that can be interpreted as a number)
24
+ scores = []
25
+ for r in results:
26
+ try:
27
+ scores.append(float(r.output))
28
+ except (TypeError, ValueError):
29
+ pass
30
+
31
+ table = Table(show_header=False, box=None, padding=(0, 2))
32
+ table.add_column("label", style="bold")
33
+ table.add_column("value")
34
+
35
+ table.add_row("Total metrics", str(total))
36
+
37
+ if scores:
38
+ avg = sum(scores) / len(scores)
39
+ table.add_row("Avg score", f"{avg:.3f}")
40
+ table.add_row("Min score", f"{min(scores):.3f}")
41
+ table.add_row("Max score", f"{max(scores):.3f}")
42
+
43
+ total_runtime = sum(r.runtime for r in results)
44
+ table.add_row("Total runtime", f"{total_runtime} ms")
45
+
46
+ self.console.print(Panel(table, title="Summary", border_style="blue"))
@@ -0,0 +1,5 @@
1
+ """Storage module for run history."""
2
+
3
+ from fi.cli.storage.run_history import RunHistory, RunRecord
4
+
5
+ __all__ = ["RunHistory", "RunRecord"]