agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,391 @@
1
+ import re
2
+ from typing import Any, Dict, List, Optional
3
+
4
+ from ..base_metric import BaseMetric
5
+ from ...types import TextMetricInput
6
+ import requests
7
+
8
+
9
+ class Regex(BaseMetric[TextMetricInput]):
10
+ """Checks if a regex pattern is found in the response text."""
11
+
12
+ @property
13
+ def metric_name(self) -> str:
14
+ return "regex"
15
+
16
+ def __init__(self, config: Optional[Dict[str, Any]] = None) -> None:
17
+ super().__init__(config)
18
+ self.pattern = self.config.get("pattern")
19
+ if not self.pattern:
20
+ raise ValueError("Regex metric requires a 'pattern' in its config.")
21
+
22
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
23
+ match = re.search(self.pattern, inputs.response)
24
+ if match:
25
+ return {
26
+ "output": 1.0, # Using 1.0 for success (True)
27
+ "reason": f"Regex pattern '{self.pattern}' found in response.",
28
+ }
29
+ else:
30
+ return {
31
+ "output": 0.0, # Using 0.0 for failure (False)
32
+ "reason": f"Regex pattern '{self.pattern}' not found in response.",
33
+ }
34
+
35
+
36
+ class Contains(BaseMetric[TextMetricInput]):
37
+ @property
38
+ def metric_name(self) -> str:
39
+ return "contains"
40
+
41
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
42
+ super().__init__(config)
43
+ self.keyword = self.config.get("keyword")
44
+ self.case_sensitive = self.config.get("case_sensitive", False)
45
+ if not self.keyword:
46
+ raise ValueError("Contains metric requires a 'keyword' config.")
47
+
48
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
49
+ text, kw = (
50
+ (inputs.response, self.keyword)
51
+ if self.case_sensitive
52
+ else (inputs.response.lower(), self.keyword.lower())
53
+ )
54
+ is_present = kw in text
55
+ return {
56
+ "output": 1.0 if is_present else 0.0,
57
+ "reason": f"Keyword '{self.keyword}' found"
58
+ if is_present
59
+ else f"Keyword '{self.keyword}' not found",
60
+ }
61
+
62
+
63
+ class _BaseContainsKeywords(BaseMetric[TextMetricInput]):
64
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
65
+ super().__init__(config)
66
+ self.keywords = self.config.get("keywords")
67
+ self.case_sensitive = self.config.get("case_sensitive", False)
68
+ if not self.keywords or not isinstance(self.keywords, list):
69
+ raise ValueError(
70
+ f"{self.metric_name} metric requires a 'keywords' list in config."
71
+ )
72
+
73
+ def _get_found_keywords(self, text: str) -> List[str]:
74
+ text_to_check = text if self.case_sensitive else text.lower()
75
+ found = [
76
+ kw
77
+ for kw in self.keywords
78
+ if (kw if self.case_sensitive else kw.lower()) in text_to_check
79
+ ]
80
+ return found
81
+
82
+
83
+ class ContainsAll(_BaseContainsKeywords):
84
+ @property
85
+ def metric_name(self) -> str:
86
+ return "contains_all"
87
+
88
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
89
+ found = self._get_found_keywords(inputs.response)
90
+ if len(found) == len(self.keywords):
91
+ return {
92
+ "output": 1.0,
93
+ "reason": f"All {len(self.keywords)} keywords found.",
94
+ }
95
+ missing = [kw for kw in self.keywords if kw not in found]
96
+ return {"output": 0.0, "reason": f"Missing keywords: {', '.join(missing)}"}
97
+
98
+
99
+ class ContainsAny(_BaseContainsKeywords):
100
+ """Checks if the response text contains any of the provided keywords."""
101
+
102
+ @property
103
+ def metric_name(self) -> str:
104
+ return "contains_any"
105
+
106
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
107
+ found_keywords = self._get_found_keywords(inputs.response)
108
+ if found_keywords:
109
+ return {
110
+ "output": 1.0,
111
+ "reason": f"Found keywords: {', '.join(found_keywords)}",
112
+ }
113
+ return {"output": 0.0, "reason": "No keywords found in response."}
114
+
115
+
116
+ class ContainsNone(_BaseContainsKeywords):
117
+ """Checks if the response text contains none of the provided keywords."""
118
+
119
+ @property
120
+ def metric_name(self) -> str:
121
+ return "contains_none"
122
+
123
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
124
+ found_keywords = self._get_found_keywords(inputs.response)
125
+ if not found_keywords:
126
+ return {"output": 1.0, "reason": "No forbidden keywords found."}
127
+ return {
128
+ "output": 0.0,
129
+ "reason": f"Found forbidden keywords: {', '.join(found_keywords)}",
130
+ }
131
+
132
+
133
+ def _standardize_url(url: str) -> str:
134
+ if url.startswith("http://") or url.startswith("https://"):
135
+ return url
136
+ return f"http://{url}"
137
+
138
+
139
+ class OneLine(BaseMetric[TextMetricInput]):
140
+ """Checks if the text is a single line."""
141
+
142
+ @property
143
+ def metric_name(self) -> str:
144
+ return "one_line"
145
+
146
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
147
+ is_one_line = "\n" not in inputs.response.strip()
148
+ return {
149
+ "output": 1.0 if is_one_line else 0.0,
150
+ "reason": "Response is a single line."
151
+ if is_one_line
152
+ else "Response contains multiple lines.",
153
+ }
154
+
155
+
156
+ class ContainsEmail(Regex):
157
+ """Checks if the text contains an email address."""
158
+
159
+ @property
160
+ def metric_name(self) -> str:
161
+ return "contains_email"
162
+
163
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
164
+ # Pass the specific regex pattern to the parent Regex class
165
+ super().__init__({"pattern": r"[a-zA-Z0-9_.+-]+@[a-zA-Z0-9-]+\.[a-zA-Z0-9-.]+"})
166
+
167
+
168
+ class IsEmail(Regex):
169
+ """Checks if the entire text is a valid email address."""
170
+
171
+ @property
172
+ def metric_name(self) -> str:
173
+ return "is_email"
174
+
175
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
176
+ super().__init__(
177
+ {"pattern": r"^[a-zA-Z0-9_.+-]+@[a-zA-Z0-9-]+\.[a-zA-Z0-9-.]+$"}
178
+ )
179
+
180
+
181
+ class ContainsLink(Regex):
182
+ """Checks if the text contains a link."""
183
+
184
+ @property
185
+ def metric_name(self) -> str:
186
+ return "contains_link"
187
+
188
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
189
+ super().__init__({"pattern": r"(?!.*@)(?:https?://)?(?:www\.)?\S+\.\S+"})
190
+
191
+
192
+ class ContainsValidLink(BaseMetric[TextMetricInput]):
193
+ """Checks if the text contains a link that returns a 2xx status code."""
194
+
195
+ @property
196
+ def metric_name(self) -> str:
197
+ return "contains_valid_link"
198
+
199
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
200
+ pattern = r"(?!.*@)(?:https?://)?(?:www\.)?\S+\.\S+"
201
+ match = re.search(pattern, inputs.response)
202
+ if not match:
203
+ return {"output": 0.0, "reason": "No link found in response."}
204
+
205
+ url = _standardize_url(match.group(0))
206
+ try:
207
+ response = requests.head(url, timeout=5)
208
+ if 200 <= response.status_code < 300:
209
+ return {
210
+ "output": 1.0,
211
+ "reason": f"Valid link '{url}' found (Status: {response.status_code})",
212
+ }
213
+ else:
214
+ return {
215
+ "output": 0.0,
216
+ "reason": f"Invalid link '{url}' found (Status: {response.status_code})",
217
+ }
218
+ except requests.RequestException as e:
219
+ return {
220
+ "output": 0.0,
221
+ "reason": f"Unreachable link '{url}' found. Error: {e}",
222
+ }
223
+
224
+
225
+ class Equals(BaseMetric[TextMetricInput]):
226
+ """Checks if the response text exactly matches the expected text."""
227
+
228
+ @property
229
+ def metric_name(self) -> str:
230
+ return "equals"
231
+
232
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
233
+ super().__init__(config)
234
+ self.case_sensitive = self.config.get("case_sensitive", False)
235
+
236
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
237
+ if inputs.expected_response is None:
238
+ raise ValueError("Equals metric requires 'expected_response' to be provided.")
239
+ if not isinstance(inputs.expected_response, str):
240
+ raise TypeError("Equals metric requires 'expected_response' to be a string.")
241
+ resp, expected = (
242
+ (inputs.response, inputs.expected_response)
243
+ if self.case_sensitive
244
+ else (inputs.response.lower(), inputs.expected_response.lower())
245
+ )
246
+ return {
247
+ "output": 1.0 if resp == expected else 0.0,
248
+ "reason": "Response matches expected text."
249
+ if resp == expected
250
+ else "Response does not match.",
251
+ }
252
+
253
+
254
+ class StartsWith(BaseMetric[TextMetricInput]):
255
+ """Checks if the response text starts with the expected text."""
256
+
257
+ @property
258
+ def metric_name(self) -> str:
259
+ return "starts_with"
260
+
261
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
262
+ super().__init__(config)
263
+ self.case_sensitive = self.config.get("case_sensitive", False)
264
+
265
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
266
+ if inputs.expected_response is None:
267
+ raise ValueError("StartsWith metric requires 'expected_response' to be provided.")
268
+ if not isinstance(inputs.expected_response, str):
269
+ raise TypeError("StartsWith requires 'expected_response' to be a string.")
270
+ resp, prefix = (
271
+ (inputs.response, inputs.expected_response)
272
+ if self.case_sensitive
273
+ else (inputs.response.lower(), inputs.expected_response.lower())
274
+ )
275
+ starts = resp.startswith(prefix)
276
+ return {
277
+ "output": 1.0 if starts else 0.0,
278
+ "reason": f"Response starts with '{inputs.expected_response}'."
279
+ if starts
280
+ else f"Response does not start with '{inputs.expected_response}'.",
281
+ }
282
+
283
+
284
+ class EndsWith(BaseMetric[TextMetricInput]):
285
+ """Checks if the response text ends with the expected text."""
286
+
287
+ @property
288
+ def metric_name(self) -> str:
289
+ return "ends_with"
290
+
291
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
292
+ super().__init__(config)
293
+ self.case_sensitive = self.config.get("case_sensitive", False)
294
+
295
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
296
+ if inputs.expected_response is None:
297
+ raise ValueError("EndsWith metric requires 'expected_response' to be provided.")
298
+ if not isinstance(inputs.expected_response, str):
299
+ raise TypeError("EndsWith requires 'expected_response' to be a string.")
300
+ resp, suffix = (
301
+ (inputs.response, inputs.expected_response)
302
+ if self.case_sensitive
303
+ else (inputs.response.lower(), inputs.expected_response.lower())
304
+ )
305
+ ends = resp.endswith(suffix)
306
+ return {
307
+ "output": 1.0 if ends else 0.0,
308
+ "reason": f"Response ends with '{inputs.expected_response}'."
309
+ if ends
310
+ else f"Response does not end with '{inputs.expected_response}'.",
311
+ }
312
+
313
+
314
+ # --- Length Metrics ---
315
+
316
+
317
+ class LengthLessThan(BaseMetric[TextMetricInput]):
318
+ """Checks if text length is less than a max_length."""
319
+
320
+ @property
321
+ def metric_name(self) -> str:
322
+ return "length_less_than"
323
+
324
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
325
+ super().__init__(config)
326
+ self.max_length = self.config.get("max_length")
327
+ if not isinstance(self.max_length, int):
328
+ raise ValueError(
329
+ "LengthLessThan metric requires an integer 'max_length' config."
330
+ )
331
+
332
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
333
+ is_less = len(inputs.response) < self.max_length
334
+ return {
335
+ "output": 1.0 if is_less else 0.0,
336
+ "reason": f"Length {len(inputs.response)} < {self.max_length}"
337
+ if is_less
338
+ else f"Length {len(inputs.response)} >= {self.max_length}",
339
+ }
340
+
341
+
342
+ class LengthGreaterThan(BaseMetric[TextMetricInput]):
343
+ """Checks if text length is greater than a min_length."""
344
+
345
+ @property
346
+ def metric_name(self) -> str:
347
+ return "length_greater_than"
348
+
349
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
350
+ super().__init__(config)
351
+ self.min_length = self.config.get("min_length")
352
+ if not isinstance(self.min_length, int):
353
+ raise ValueError(
354
+ "LengthGreaterThan metric requires an integer 'min_length' config."
355
+ )
356
+
357
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
358
+ is_greater = len(inputs.response) > self.min_length
359
+ return {
360
+ "output": 1.0 if is_greater else 0.0,
361
+ "reason": f"Length {len(inputs.response)} > {self.min_length}"
362
+ if is_greater
363
+ else f"Length {len(inputs.response)} <= {self.min_length}",
364
+ }
365
+
366
+
367
+ class LengthBetween(BaseMetric[TextMetricInput]):
368
+ """Checks if text length is between a min_length and max_length."""
369
+
370
+ @property
371
+ def metric_name(self) -> str:
372
+ return "length_between"
373
+
374
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
375
+ super().__init__(config)
376
+ self.min_length = self.config.get("min_length")
377
+ self.max_length = self.config.get("max_length")
378
+ if not isinstance(self.min_length, int) or not isinstance(self.max_length, int):
379
+ raise ValueError(
380
+ "LengthBetween requires integer 'min_length' and 'max_length' configs."
381
+ )
382
+
383
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
384
+ length = len(inputs.response)
385
+ is_between = self.min_length <= length <= self.max_length
386
+ reason = (
387
+ f"Length {length} is between [{self.min_length}, {self.max_length}]"
388
+ if is_between
389
+ else f"Length {length} is not between [{self.min_length}, {self.max_length}]"
390
+ )
391
+ return {"output": 1.0 if is_between else 0.0, "reason": reason}
@@ -0,0 +1,17 @@
1
+ from .custom_judge.metric import CustomLLMJudge
2
+ from .types import (
3
+ CustomInput,
4
+ BaseLLMJudgeInput,
5
+ LLMFewShotExample,
6
+ LLMMessage,
7
+ DefaultJudgeOutput,
8
+ )
9
+
10
+ __all__ = [
11
+ "CustomLLMJudge",
12
+ "CustomInput",
13
+ "BaseLLMJudgeInput",
14
+ "LLMFewShotExample",
15
+ "LLMMessage",
16
+ "DefaultJudgeOutput",
17
+ ]
@@ -0,0 +1,112 @@
1
+ import json
2
+ from typing import Any, Dict, List, Type
3
+ from pydantic import BaseModel
4
+ from jinja2 import Environment, BaseLoader
5
+
6
+ from ...base_llm_metric import BaseLLMJudgeMetric
7
+ from ..types import CustomInput, DefaultJudgeOutput
8
+ from ....llm.base_llm_provider import LLMProvider
9
+ from .prompts import DEFAULT_USER_PROMPT_TEMPLATE
10
+
11
+
12
+ class CustomLLMJudge(BaseLLMJudgeMetric[CustomInput]):
13
+ """
14
+ A smart, user-configurable LLM-as-a-judge metric that prioritizes ease of use.
15
+
16
+ For the most common use cases, the user only needs to provide their
17
+ grading criteria. The judge provides sensible defaults for the prompt
18
+ template and output format, which can be optionally overridden for
19
+ advanced customization.
20
+ """
21
+
22
+ @property
23
+ def metric_name(self) -> str:
24
+ return self.config.get("name", "custom_llm_judge")
25
+
26
+ def __init__(self, provider: LLMProvider, config: Dict[str, Any], **litellm_kwargs):
27
+ # The ONLY required key is now 'grading_criteria'
28
+ if "grading_criteria" not in config:
29
+ raise ValueError(
30
+ "CustomLLMJudge config must contain a 'grading_criteria' key."
31
+ )
32
+
33
+ super().__init__(provider, config, **litellm_kwargs)
34
+
35
+ # Explicitly set the input model, as this class is generic
36
+ self.input_model = CustomInput
37
+
38
+ # Smartly decide which Pydantic model to uset
39
+ self._output_model = DefaultJudgeOutput
40
+
41
+ @property
42
+ def output_pydantic_model(self) -> Type[BaseModel]:
43
+ return self._output_model
44
+
45
+ # Keys whose values are media URLs that should be sent as content parts
46
+ _IMAGE_KEYS = {"image_url", "input_image_url", "output_image_url", "image"}
47
+ _AUDIO_KEYS = {"audio_url", "input_audio_url", "audio"}
48
+
49
+ def _create_prompt_messages(self, inputs: CustomInput) -> List[Dict[str, Any]]:
50
+ jinja_env = Environment(loader=BaseLoader())
51
+ jinja_env.filters["tojson"] = json.dumps
52
+
53
+ # Use the user-provided template if it exists, otherwise use the default
54
+ template_str = self.config.get(
55
+ "user_prompt_template", DEFAULT_USER_PROMPT_TEMPLATE
56
+ )
57
+ template = jinja_env.from_string(template_str)
58
+
59
+ render_context = {
60
+ "grading_criteria": self.config["grading_criteria"],
61
+ "few_shot_examples": self.config.get("few_shot_examples", []),
62
+ "task_input": inputs.model_dump(),
63
+ }
64
+
65
+ user_prompt = template.render(render_context)
66
+
67
+ system_prompt = self.config.get(
68
+ "system_prompt",
69
+ "You are an expert AI evaluator. Follow the user's instructions and output format precisely.",
70
+ )
71
+
72
+ # Detect multimodal inputs and build content parts
73
+ user_content = self._build_content(user_prompt, inputs.model_dump())
74
+
75
+ return [
76
+ {"role": "system", "content": system_prompt},
77
+ {"role": "user", "content": user_content},
78
+ ]
79
+
80
+ def _build_content(self, text: str, input_data: Dict[str, Any]):
81
+ """Build message content — plain string or list of content parts with media."""
82
+ media_parts = []
83
+
84
+ for key, value in input_data.items():
85
+ if not value or not isinstance(value, str):
86
+ continue
87
+ if key in self._IMAGE_KEYS:
88
+ media_parts.append({
89
+ "type": "image_url",
90
+ "image_url": {"url": value},
91
+ })
92
+ elif key in self._AUDIO_KEYS:
93
+ # LiteLLM translates image_url type to the correct provider
94
+ # format for audio URLs too (Gemini, OpenAI, etc.)
95
+ media_parts.append({
96
+ "type": "image_url",
97
+ "image_url": {"url": value},
98
+ })
99
+
100
+ if not media_parts:
101
+ return text
102
+
103
+ return [{"type": "text", "text": text}] + media_parts
104
+
105
+ def _normalize_score(self, parsed_output: BaseModel) -> Dict[str, Any]:
106
+ """Normalizes the score from the validated Pydantic output."""
107
+ output_dict = parsed_output.model_dump()
108
+
109
+ # Prioritize finding a "score" field, as it's our default
110
+ score_val = output_dict.get("score", 1.0)
111
+
112
+ return {"output": float(score_val), "reason": json.dumps(output_dict, indent=2)}
@@ -0,0 +1,26 @@
1
+ DEFAULT_USER_PROMPT_TEMPLATE = """
2
+ ### GRADING CRITERIA ###
3
+ {{ grading_criteria }}
4
+
5
+ ### OUTPUT FORMAT ###
6
+ You MUST return a valid JSON object with two keys: "score" (a float between 0.0 and 1.0) and "reason" (a brief explanation of your score).
7
+
8
+ {% if few_shot_examples %}
9
+ ### EXAMPLES ###
10
+ {% for example in few_shot_examples -%}
11
+ ---
12
+ INPUT:
13
+ {{ example.inputs | tojson(indent=2) }}
14
+
15
+ EXPECTED JUDGEMENT:
16
+ {{ example.output }}
17
+ ---
18
+ {% endfor %}
19
+ {% endif %}
20
+
21
+ ### TASK ###
22
+ Based on the grading criteria, please evaluate the following input.
23
+
24
+ ### INPUT ###
25
+ {{ task_input | tojson(indent=2) }}
26
+ """
@@ -0,0 +1,48 @@
1
+ from typing import Any, Dict, List, Optional
2
+ from pydantic import BaseModel, Field
3
+ from ...types import BaseMetricInput
4
+
5
+
6
+ class BaseLLMJudgeInput(BaseMetricInput):
7
+ pass
8
+
9
+
10
+ class LLMFewShotExample(BaseModel):
11
+ inputs: Dict[str, Any] = Field(
12
+ ..., description="A dictionary representing an input model"
13
+ )
14
+ output: str = Field(
15
+ ...,
16
+ description="The ideal JSON string the judge LLM should produce for these inputs.",
17
+ )
18
+
19
+
20
+ class CustomInput(BaseLLMJudgeInput):
21
+ """A flexible input model for the CustomLLMJudge that allows any field."""
22
+
23
+ class Config:
24
+ extra = "allow"
25
+
26
+
27
+ class DefaultJudgeOutput(BaseModel):
28
+ """The default output format for a custom judge."""
29
+
30
+ score: float = Field(
31
+ ...,
32
+ ge=0.0,
33
+ le=1.0,
34
+ description="The normalized evaluation score from 0.0 to 1.0.",
35
+ )
36
+ reason: str = Field(..., description="A brief explanation of the score.")
37
+
38
+
39
+ class LLMMessage(BaseModel):
40
+ role: str
41
+ content: str
42
+ name: Optional[str]
43
+ function_call: Optional[str]
44
+ tool_call_id: Optional[str]
45
+
46
+
47
+ class ConversationInput(BaseLLMJudgeInput):
48
+ messages: List[LLMMessage]