agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,319 @@
1
+ """
2
+ evaluate() — the unified entrypoint for all evaluations.
3
+
4
+ Usage:
5
+ from fi.evals import evaluate
6
+
7
+ # Local metric (auto-detected)
8
+ result = evaluate("contains", output="hello world", keyword="hello")
9
+
10
+ # Cloud template (turing model → auto-routes to Turing)
11
+ result = evaluate("toxicity", output="hello world", model="turing_flash")
12
+
13
+ # Custom prompt on any LLM (use engine="llm", not "turing")
14
+ result = evaluate(
15
+ prompt="Rate the clarity of: {output}",
16
+ output="ML is a subset of AI.",
17
+ engine="llm",
18
+ model="gemini/gemini-2.0-flash",
19
+ )
20
+
21
+ # LLM-augmented: local heuristic first, then LLM refines
22
+ result = evaluate(
23
+ "faithfulness",
24
+ output="The capital of France is Paris.",
25
+ context="Paris is the capital of France.",
26
+ model="gemini/gemini-2.5-flash",
27
+ augment=True,
28
+ )
29
+
30
+ # Multiple evals
31
+ results = evaluate(
32
+ ["toxicity", "factual_accuracy"],
33
+ output="Paris is the capital of France",
34
+ model="turing_flash",
35
+ )
36
+ """
37
+
38
+ import warnings
39
+ from typing import Any, Dict, List, Optional, Union
40
+
41
+ from .registry import resolve_engine as _resolve_engine, is_turing_model
42
+ from .result import BatchResult, EvalResult
43
+ from .engines import LocalEngine, TuringEngine, LLMEngine, Engine
44
+
45
+
46
+ def evaluate(
47
+ eval_name: Optional[Union[str, List[str]]] = None,
48
+ *,
49
+ prompt: Optional[str] = None,
50
+ engine: Optional[str] = None,
51
+ model: Optional[str] = None,
52
+ augment: Optional[bool] = None,
53
+ config: Optional[Dict[str, Any]] = None,
54
+ feedback_store: Optional[Any] = None,
55
+ generate_prompt: bool = False,
56
+ # Turing credentials (optional overrides)
57
+ fi_api_key: Optional[str] = None,
58
+ fi_secret_key: Optional[str] = None,
59
+ fi_base_url: Optional[str] = None,
60
+ **inputs,
61
+ ) -> Union[EvalResult, BatchResult]:
62
+ """Run one or more evaluations with automatic engine routing.
63
+
64
+ Args:
65
+ eval_name: Metric/template name or list of names. Can be None when
66
+ using a custom prompt.
67
+ prompt: Custom evaluation prompt with {output}/{context}/{input}
68
+ placeholders. Requires explicit engine or a model hint.
69
+ Note: custom prompts only work with engine='llm'.
70
+ engine: Force a specific engine — "local", "turing", or "llm".
71
+ model: Model to use. Turing models (e.g. "turing_flash") auto-route
72
+ to the Turing engine. Other model strings (e.g.
73
+ "gemini/gemini-2.0-flash") auto-route to the LLM engine.
74
+ augment: When True, run the local heuristic first, then pass its
75
+ scores + reasoning to the LLM (specified by model=) for
76
+ refinement. Requires model= and a metric that supports
77
+ LLM augmentation (supports_llm_judge = True).
78
+ config: Optional metric/judge config dict.
79
+ feedback_store: Optional FeedbackStore for retrieving few-shot
80
+ examples from developer feedback. When provided with
81
+ augment=True, similar past feedback is injected into
82
+ the LLM judge prompt.
83
+ generate_prompt: When True, treat `prompt` as a short description
84
+ and auto-generate detailed grading criteria via LLM.
85
+ Requires model= and prompt=.
86
+ fi_api_key: Override FI_API_KEY for Turing engine.
87
+ fi_secret_key: Override FI_SECRET_KEY for Turing engine.
88
+ fi_base_url: Override FI_BASE_URL for Turing engine.
89
+ **inputs: Evaluation inputs (output, context, input, keyword, …).
90
+
91
+ Returns:
92
+ EvalResult for a single eval, BatchResult for multiple.
93
+ """
94
+ # --- Batch case: list of eval names --------------------------------
95
+ if isinstance(eval_name, list):
96
+ results = []
97
+ for name in eval_name:
98
+ r = evaluate(
99
+ name,
100
+ prompt=prompt,
101
+ engine=engine,
102
+ model=model,
103
+ augment=augment,
104
+ config=config,
105
+ feedback_store=feedback_store,
106
+ generate_prompt=generate_prompt,
107
+ fi_api_key=fi_api_key,
108
+ fi_secret_key=fi_secret_key,
109
+ fi_base_url=fi_base_url,
110
+ **inputs,
111
+ )
112
+ results.append(r)
113
+ return BatchResult(results=results)
114
+
115
+ # --- Single eval ---------------------------------------------------
116
+ # Auto-generate grading criteria from a short description
117
+ if generate_prompt and prompt:
118
+ if not model:
119
+ raise ValueError(
120
+ "generate_prompt=True requires a model= parameter "
121
+ "(e.g. model='gemini/gemini-2.5-flash')."
122
+ )
123
+ from .prompt_generator import generate_grading_criteria
124
+ prompt = generate_grading_criteria(prompt, model, inputs)
125
+
126
+ # Custom prompt with no eval_name
127
+ effective_name = eval_name or "custom_prompt"
128
+
129
+ # When augment=True, force local engine for initial run — model is for LLM step
130
+ resolved_engine = _resolve_engine(
131
+ eval_name,
132
+ model=None if augment else model,
133
+ prompt=prompt,
134
+ engine=engine,
135
+ )
136
+
137
+ if resolved_engine is None:
138
+ raise ValueError(
139
+ f"Cannot auto-detect engine for '{eval_name}'. "
140
+ "Specify engine='local', engine='turing', or engine='llm', "
141
+ "or provide a model (turing models → turing, others → llm)."
142
+ )
143
+
144
+ eng = _get_engine(
145
+ resolved_engine,
146
+ fi_api_key=fi_api_key,
147
+ fi_secret_key=fi_secret_key,
148
+ fi_base_url=fi_base_url,
149
+ )
150
+
151
+ # Run with OTEL span if tracing is enabled
152
+ result = _run_with_tracing(
153
+ eng, effective_name, inputs,
154
+ model=model, prompt=prompt, config=config,
155
+ engine_type=resolved_engine,
156
+ )
157
+
158
+ # LLM augmentation: only when explicitly requested via augment=True
159
+ if augment:
160
+ result = _augment_with_llm(
161
+ result,
162
+ effective_name=effective_name,
163
+ inputs=inputs,
164
+ model=model,
165
+ resolved_engine=resolved_engine,
166
+ feedback_store=feedback_store,
167
+ )
168
+
169
+ return result
170
+
171
+
172
+ _ENGINE_FACTORIES = {
173
+ "local": lambda **_: LocalEngine(),
174
+ "turing": lambda **kw: TuringEngine(
175
+ fi_api_key=kw.get("fi_api_key"),
176
+ fi_secret_key=kw.get("fi_secret_key"),
177
+ fi_base_url=kw.get("fi_base_url"),
178
+ ),
179
+ "llm": lambda **_: LLMEngine(),
180
+ }
181
+
182
+
183
+ def _get_engine(
184
+ engine_type: str,
185
+ *,
186
+ fi_api_key: Optional[str] = None,
187
+ fi_secret_key: Optional[str] = None,
188
+ fi_base_url: Optional[str] = None,
189
+ ) -> Engine:
190
+ factory = _ENGINE_FACTORIES.get(engine_type.lower())
191
+ if factory is None:
192
+ raise ValueError(
193
+ f"Unknown engine: '{engine_type}'. "
194
+ f"Use one of: {', '.join(sorted(_ENGINE_FACTORIES))}."
195
+ )
196
+ return factory(fi_api_key=fi_api_key, fi_secret_key=fi_secret_key, fi_base_url=fi_base_url)
197
+
198
+
199
+ def _run_with_tracing(
200
+ eng: Engine,
201
+ eval_name: str,
202
+ inputs: Dict[str, Any],
203
+ *,
204
+ model: Optional[str] = None,
205
+ prompt: Optional[str] = None,
206
+ config: Optional[Dict[str, Any]] = None,
207
+ engine_type: str = "",
208
+ ) -> EvalResult:
209
+ """Run an engine with optional OTEL tracing."""
210
+ try:
211
+ from fi.evals.otel.enrichment import (
212
+ is_auto_enrichment_enabled,
213
+ create_evaluation_span,
214
+ enrich_span_with_evaluation,
215
+ )
216
+ if not is_auto_enrichment_enabled():
217
+ raise ImportError # fall through to untraced path
218
+
219
+ with create_evaluation_span(eval_name) as span:
220
+ result = eng.run(eval_name, inputs, model=model, prompt=prompt, config=config)
221
+ # Enrich the span with the result
222
+ if hasattr(span, "set_attribute"):
223
+ span.set_attribute("gen_ai.span.kind", "EVALUATOR")
224
+ span.set_attribute("gen_ai.evaluation.name", eval_name)
225
+ enrich_span_with_evaluation(
226
+ metric_name=result.eval_name,
227
+ score=result.score if result.score is not None else 0.0,
228
+ reason=result.reason,
229
+ latency_ms=result.latency_ms,
230
+ span=span if hasattr(span, "set_attribute") else None,
231
+ )
232
+ return result
233
+ except ImportError:
234
+ pass
235
+
236
+ return eng.run(eval_name, inputs, model=model, prompt=prompt, config=config)
237
+
238
+
239
+ def _augment_with_llm(
240
+ result: EvalResult,
241
+ *,
242
+ effective_name: str,
243
+ inputs: Dict[str, Any],
244
+ model: Optional[str],
245
+ resolved_engine: str,
246
+ feedback_store: Optional[Any] = None,
247
+ ) -> EvalResult:
248
+ """Augment a local heuristic result with LLM judgment.
249
+
250
+ Called only when augment=True. Validates preconditions and raises
251
+ clear errors instead of silently skipping.
252
+ """
253
+ if not model:
254
+ raise ValueError(
255
+ "augment=True requires a model= parameter "
256
+ "(e.g. model='gemini/gemini-2.5-flash')."
257
+ )
258
+ if is_turing_model(model):
259
+ raise ValueError(
260
+ f"augment=True is not compatible with Turing models (got '{model}'). "
261
+ "Use a LiteLLM model string like 'gemini/gemini-2.5-flash'."
262
+ )
263
+ if resolved_engine != "local":
264
+ raise ValueError(
265
+ f"augment=True only works with local metrics, but engine resolved "
266
+ f"to '{resolved_engine}'. Remove engine= or set engine='local'."
267
+ )
268
+
269
+ from ..local.registry import get_registry
270
+ metric_cls = get_registry().get(effective_name)
271
+
272
+ if metric_cls is None or not getattr(metric_cls, "supports_llm_judge", False):
273
+ raise ValueError(
274
+ f"Metric '{effective_name}' does not support LLM augmentation "
275
+ f"(supports_llm_judge is False). Only judgment metrics like "
276
+ f"faithfulness, hallucination_score, task_completion, etc. can be augmented."
277
+ )
278
+
279
+ if result.status != "completed":
280
+ result.metadata["engine"] = "local"
281
+ return result
282
+
283
+ description = getattr(metric_cls, "judge_description", "") or ""
284
+
285
+ from .judge_prompt import build_judge_prompt
286
+ judge_prompt = build_judge_prompt(effective_name, description, inputs, result)
287
+
288
+ # Retrieve feedback-based few-shot examples if available
289
+ llm_config = None
290
+ try:
291
+ from ..feedback.hooks import retrieve_feedback_config
292
+ llm_config = retrieve_feedback_config(
293
+ metric_name=effective_name,
294
+ inputs=inputs,
295
+ store=feedback_store,
296
+ )
297
+ except ImportError:
298
+ pass # feedback module not installed / not configured
299
+
300
+ llm_eng = LLMEngine()
301
+ try:
302
+ augmented = llm_eng.run(
303
+ effective_name, inputs,
304
+ model=model, prompt=judge_prompt, config=llm_config,
305
+ )
306
+ augmented.metadata["engine"] = "local+llm"
307
+ if llm_config and llm_config.get("few_shot_examples"):
308
+ augmented.metadata["feedback_examples_used"] = len(llm_config["few_shot_examples"])
309
+ return augmented
310
+ except Exception as exc:
311
+ warnings.warn(
312
+ f"LLM augmentation failed for '{effective_name}', "
313
+ f"falling back to local heuristic result: {exc}",
314
+ RuntimeWarning,
315
+ stacklevel=3,
316
+ )
317
+ result.metadata["engine"] = "local"
318
+ result.metadata["augment_error"] = str(exc)
319
+ return result
@@ -0,0 +1,90 @@
1
+ """
2
+ Generic LLM judge prompt builder for augmenting local metric results.
3
+
4
+ When a local metric supports LLM augmentation (supports_llm_judge = True)
5
+ and the user passes augment=True, the local heuristic runs first, then
6
+ this module builds a prompt that feeds the heuristic scores + reasoning
7
+ to the LLM for refinement.
8
+ """
9
+
10
+ import json
11
+ from typing import Any, Dict
12
+
13
+ from .result import EvalResult
14
+
15
+
16
+ _PROMPT_TEMPLATE = """\
17
+ You are an expert AI evaluator. Your task is to evaluate: **{metric_name}**
18
+
19
+ ## What this metric measures
20
+ {description}
21
+
22
+ ## Local analysis (heuristic pre-screening)
23
+ The following analysis was produced by a fast, deterministic heuristic.
24
+ Use it as a starting point — it may be accurate, but it cannot reason
25
+ about semantics the way you can.
26
+
27
+ Score: {local_score}
28
+ Reasoning: {local_reason}
29
+
30
+ ## Raw data
31
+ {formatted_inputs}
32
+
33
+ ## Instructions
34
+ Using the local analysis as a starting point and the raw data for verification,
35
+ provide your refined judgment.
36
+
37
+ - If the heuristic score seems correct, confirm it with your own reasoning.
38
+ - If you find the heuristic missed something or was too harsh/lenient, adjust.
39
+ - Score from 0.0 (worst) to 1.0 (best).
40
+
41
+ Return ONLY a JSON object: {{"score": <float>, "reason": "<brief explanation>"}}\
42
+ """
43
+
44
+
45
+ def _format_inputs(inputs: Dict[str, Any]) -> str:
46
+ """Format evaluation inputs for the prompt, keeping it concise."""
47
+ parts = []
48
+ for key, value in inputs.items():
49
+ if value is None:
50
+ continue
51
+ if isinstance(value, str):
52
+ display = value if len(value) <= 1000 else value[:1000] + "..."
53
+ parts.append(f"**{key}**:\n{display}")
54
+ elif isinstance(value, (list, dict)):
55
+ try:
56
+ dumped = json.dumps(value, indent=2, default=str)
57
+ if len(dumped) > 1500:
58
+ dumped = dumped[:1500] + "\n..."
59
+ parts.append(f"**{key}**:\n```json\n{dumped}\n```")
60
+ except (TypeError, ValueError):
61
+ parts.append(f"**{key}**: {str(value)[:500]}")
62
+ else:
63
+ parts.append(f"**{key}**: {value}")
64
+ return "\n\n".join(parts) if parts else "(no inputs)"
65
+
66
+
67
+ def build_judge_prompt(
68
+ metric_name: str,
69
+ description: str,
70
+ inputs: Dict[str, Any],
71
+ local_result: EvalResult,
72
+ ) -> str:
73
+ """Build an LLM judge prompt that includes local heuristic results.
74
+
75
+ Args:
76
+ metric_name: The metric being evaluated (e.g. "faithfulness").
77
+ description: What this metric measures (from metric_cls.judge_description).
78
+ inputs: The raw evaluation inputs (output, context, trajectory, etc.).
79
+ local_result: The EvalResult from the local heuristic engine.
80
+
81
+ Returns:
82
+ A formatted prompt string ready for the LLM engine.
83
+ """
84
+ return _PROMPT_TEMPLATE.format(
85
+ metric_name=metric_name,
86
+ description=description or f"Evaluate the quality of the output for '{metric_name}'.",
87
+ local_score=local_result.score if local_result.score is not None else "N/A",
88
+ local_reason=local_result.reason or "No reasoning provided.",
89
+ formatted_inputs=_format_inputs(inputs),
90
+ )
@@ -0,0 +1,83 @@
1
+ """
2
+ Auto-generate grading criteria from a short description.
3
+
4
+ Usage:
5
+ from fi.evals import evaluate
6
+
7
+ # Instead of writing a detailed rubric yourself:
8
+ result = evaluate(
9
+ prompt="product description accuracy for e-commerce images",
10
+ output="A red cotton t-shirt with v-neck",
11
+ image_url="https://example.com/tshirt.jpg",
12
+ engine="llm",
13
+ model="gemini/gemini-2.5-flash",
14
+ generate_prompt=True,
15
+ )
16
+ """
17
+
18
+ import hashlib
19
+ from typing import Any, Dict
20
+
21
+ _CACHE: Dict[str, str] = {}
22
+
23
+ _META_PROMPT = """\
24
+ You are an expert prompt engineer specializing in LLM evaluation rubrics.
25
+
26
+ Given a short description of what to evaluate, generate a detailed grading \
27
+ criteria that an LLM judge can use to score inputs on a 0.0–1.0 scale.
28
+
29
+ The criteria MUST:
30
+ - Be specific and actionable (not vague)
31
+ - Define what 1.0, 0.5, and 0.0 look like
32
+ - Reference the input fields the judge will receive: {input_keys}
33
+ - Be 4-8 sentences maximum
34
+
35
+ Description of what to evaluate:
36
+ {description}
37
+
38
+ Return ONLY the grading criteria text. No JSON, no markdown, no preamble.\
39
+ """
40
+
41
+
42
+ def generate_grading_criteria(
43
+ description: str,
44
+ model: str,
45
+ inputs: Dict[str, Any],
46
+ *,
47
+ cache: bool = True,
48
+ ) -> str:
49
+ """Generate a detailed grading criteria from a short description.
50
+
51
+ Args:
52
+ description: Short description of what to evaluate
53
+ (e.g. "product description accuracy for images").
54
+ model: LiteLLM model string (e.g. "gemini/gemini-2.5-flash").
55
+ inputs: The evaluation inputs dict — used to tell the generator
56
+ which fields the judge will receive.
57
+ cache: Cache results per (description, model) for the session.
58
+
59
+ Returns:
60
+ A detailed grading criteria string.
61
+ """
62
+ cache_key = hashlib.md5(f"{description}:{model}".encode()).hexdigest()
63
+ if cache and cache_key in _CACHE:
64
+ return _CACHE[cache_key]
65
+
66
+ input_keys = ", ".join(sorted(inputs.keys())) or "output"
67
+
68
+ prompt = _META_PROMPT.format(
69
+ description=description,
70
+ input_keys=input_keys,
71
+ )
72
+
73
+ import litellm
74
+ response = litellm.completion(
75
+ model=model,
76
+ messages=[{"role": "user", "content": prompt}],
77
+ )
78
+ criteria = response.choices[0].message.content.strip()
79
+
80
+ if cache:
81
+ _CACHE[cache_key] = criteria
82
+
83
+ return criteria
@@ -0,0 +1,57 @@
1
+ """
2
+ Unified registry — resolves eval names to engines.
3
+
4
+ Routing is purely based on user-provided kwargs:
5
+ 1. Explicit engine= → use that
6
+ 2. Turing model → "turing"
7
+ 3. Any other model → "llm"
8
+ 4. No model → "local" (default)
9
+ """
10
+
11
+ from enum import Enum
12
+ from typing import Optional
13
+
14
+
15
+ class Turing(str, Enum):
16
+ """Model options for the Turing (FutureAGI) cloud engine."""
17
+
18
+ FLASH = "turing_flash"
19
+ SMALL = "turing_small"
20
+ LARGE = "turing_large"
21
+
22
+
23
+ TURING_MODEL_PREFIXES = ("turing",)
24
+
25
+
26
+ def is_turing_model(model: Optional[str]) -> bool:
27
+ """Check if the model string indicates a Turing platform model."""
28
+ if not model:
29
+ return False
30
+ return model.lower().startswith(TURING_MODEL_PREFIXES)
31
+
32
+
33
+ def resolve_engine(
34
+ name: Optional[str] = None,
35
+ *,
36
+ model: Optional[str] = None,
37
+ prompt: Optional[str] = None,
38
+ engine: Optional[str] = None,
39
+ ) -> str:
40
+ """Auto-detect which engine based on user-provided kwargs.
41
+
42
+ Priority:
43
+ 1. Explicit engine kwarg
44
+ 2. Turing model → "turing"
45
+ 3. Any other model → "llm"
46
+ 4. No model → "local" (default, fails gracefully if metric not found)
47
+ """
48
+ if engine:
49
+ return engine
50
+
51
+ if is_turing_model(model):
52
+ return "turing"
53
+
54
+ if model:
55
+ return "llm"
56
+
57
+ return "local"
@@ -0,0 +1,55 @@
1
+ """
2
+ Unified result types for all evaluations.
3
+
4
+ EvalResult is the ONE result type returned by evaluate() and all engines.
5
+ BatchResult wraps multiple EvalResults when running several evals at once.
6
+ """
7
+
8
+ from dataclasses import dataclass, field
9
+ from typing import Any, Dict, Iterator, List, Optional
10
+
11
+
12
+ @dataclass
13
+ class EvalResult:
14
+ """The unified result type returned by evaluate() and all engines."""
15
+
16
+ eval_name: str
17
+ score: Optional[float] = None
18
+ passed: Optional[bool] = None
19
+ reason: str = ""
20
+ latency_ms: float = 0.0
21
+ status: str = "completed"
22
+ error: Optional[str] = None
23
+ metadata: Dict[str, Any] = field(default_factory=dict)
24
+
25
+ def __post_init__(self):
26
+ # Auto-derive passed from score if not explicitly set
27
+ if self.passed is None and self.score is not None:
28
+ self.passed = self.score >= 0.5
29
+
30
+
31
+ @dataclass
32
+ class BatchResult:
33
+ """Returned when multiple evals are run via evaluate()."""
34
+
35
+ results: List[EvalResult] = field(default_factory=list)
36
+
37
+ @property
38
+ def success_rate(self) -> float:
39
+ if not self.results:
40
+ return 0.0
41
+ completed = sum(1 for r in self.results if r.status == "completed")
42
+ return completed / len(self.results)
43
+
44
+ def get(self, name: str) -> Optional[EvalResult]:
45
+ """Get result by eval name."""
46
+ for r in self.results:
47
+ if r.eval_name == name:
48
+ return r
49
+ return None
50
+
51
+ def __iter__(self) -> Iterator[EvalResult]:
52
+ return iter(self.results)
53
+
54
+ def __len__(self) -> int:
55
+ return len(self.results)