agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,184 @@
1
+ """
2
+ Cloud eval registry — source of truth for eval metadata.
3
+
4
+ Fetches the full template list from `/sdk/api/v1/get-evals/` once per
5
+ (base_url, api_key) tuple and caches the result. Exposes helpers that
6
+ turn user-supplied kwargs into the exact key set the backend will accept,
7
+ so the SDK never drifts from the backend when evals are added/renamed.
8
+
9
+ Why this exists
10
+ ---------------
11
+
12
+ The old ``templates.py`` hardcoded Pydantic Input models per template
13
+ (``OutputOnly``, ``OutputWithContext``, ``OutputWithExpected``, ...). The
14
+ Turing revamp renamed/removed/replaced ~50 templates; the hardcoded
15
+ schemas drifted and 28 of 57 stopped working. Going forward, the backend
16
+ is authoritative — the SDK reads ``required_keys`` and sends only those.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import logging
21
+ import threading
22
+ from typing import Any, Dict, List, Optional, Set, Tuple
23
+
24
+ log = logging.getLogger(__name__)
25
+
26
+
27
+ # Module-level cache: {(base_url, api_key_prefix): {eval_name: info_dict}}
28
+ _CACHE: Dict[Tuple[str, str], Dict[str, Dict[str, Any]]] = {}
29
+ _CACHE_LOCK = threading.Lock()
30
+
31
+ # Aliases user kwargs can use → canonical backend keys.
32
+ # Maps are checked in order; first match wins.
33
+ _KEY_ALIASES: Dict[str, Tuple[str, ...]] = {
34
+ # Bidirectional output↔input — some evals only accept one of the two
35
+ # (e.g. prompt_injection wants `input`, toxicity wants `output`).
36
+ # Aliasing only fires when the canonical key isn't already in user_inputs,
37
+ # so we can't accidentally overwrite an explicit user value.
38
+ "output": ("output", "response", "answer", "generated", "input"),
39
+ "input": ("input", "query", "question", "prompt_input", "output"),
40
+ "context": ("context", "contexts"),
41
+ "expected": ("expected", "expected_output", "expected_response", "ground_truth"),
42
+ "expected_value": ("expected_value", "expected_output", "expected_response", "ground_truth"),
43
+ "generated_value": ("generated_value", "output", "response", "answer"),
44
+ "reference": ("reference", "expected_output", "expected_response", "ground_truth"),
45
+ "hypothesis": ("hypothesis", "output", "response"),
46
+ "text": ("text", "output", "content"),
47
+ "conversation": ("conversation", "messages"),
48
+ "prompt": ("prompt", "instructions", "system_prompt"),
49
+ "system_prompt": ("system_prompt", "prompt", "instructions"),
50
+ "image": ("image", "image_url", "input_image_url"),
51
+ "caption": ("caption", "output"),
52
+ "instruction": ("instruction", "prompt"),
53
+ "instructions": ("instructions", "prompt"),
54
+ "images": ("images", "image_urls", "input_image_urls"),
55
+ "input_pdf": ("input_pdf", "pdf"),
56
+ "json_content": ("json_content", "json", "expected_output"),
57
+ "input_audio": ("input_audio", "audio"),
58
+ "audio": ("audio", "input_audio"),
59
+ "generated_audio": ("generated_audio", "audio", "output"),
60
+ "generated_transcript": ("generated_transcript", "transcript", "output"),
61
+ }
62
+
63
+
64
+ def _cache_key(base_url: str, api_key: Optional[str]) -> Tuple[str, str]:
65
+ return (base_url.rstrip("/"), (api_key or "")[:12])
66
+
67
+
68
+ def load_registry(
69
+ base_url: str,
70
+ api_key: Optional[str],
71
+ secret_key: Optional[str],
72
+ *,
73
+ force_refresh: bool = False,
74
+ ) -> Dict[str, Dict[str, Any]]:
75
+ """Fetch and cache the eval template list. Returns {eval_name: info}."""
76
+ key = _cache_key(base_url, api_key)
77
+ if not force_refresh:
78
+ with _CACHE_LOCK:
79
+ cached = _CACHE.get(key)
80
+ if cached is not None:
81
+ return cached
82
+
83
+ # Imported lazily to avoid a circular import at module load time.
84
+ import requests
85
+
86
+ url = f"{base_url.rstrip('/')}/sdk/api/v1/get-evals/"
87
+ headers = {}
88
+ if api_key:
89
+ headers["X-Api-Key"] = api_key
90
+ if secret_key:
91
+ headers["X-Secret-Key"] = secret_key
92
+
93
+ try:
94
+ resp = requests.get(url, headers=headers, timeout=30)
95
+ resp.raise_for_status()
96
+ body = resp.json()
97
+ except Exception as exc:
98
+ log.warning("Failed to load cloud eval registry from %s: %s", url, exc)
99
+ return {}
100
+
101
+ items = body.get("result") or []
102
+ by_name: Dict[str, Dict[str, Any]] = {}
103
+ for item in items:
104
+ name = item.get("name")
105
+ if name:
106
+ by_name[name] = item
107
+
108
+ with _CACHE_LOCK:
109
+ _CACHE[key] = by_name
110
+ log.debug("Loaded %d cloud eval templates from %s", len(by_name), url)
111
+ return by_name
112
+
113
+
114
+ def get_template_info(
115
+ name: str,
116
+ *,
117
+ base_url: str,
118
+ api_key: Optional[str] = None,
119
+ secret_key: Optional[str] = None,
120
+ ) -> Optional[Dict[str, Any]]:
121
+ """Return the backend's config block for a given eval, or None."""
122
+ reg = load_registry(base_url, api_key, secret_key)
123
+ return reg.get(name)
124
+
125
+
126
+ def get_required_keys(
127
+ name: str,
128
+ *,
129
+ base_url: str,
130
+ api_key: Optional[str] = None,
131
+ secret_key: Optional[str] = None,
132
+ ) -> List[str]:
133
+ """List the keys the backend requires for this eval. Empty list if unknown."""
134
+ info = get_template_info(name, base_url=base_url, api_key=api_key, secret_key=secret_key)
135
+ if not info:
136
+ return []
137
+ return list(info.get("config", {}).get("required_keys", []) or [])
138
+
139
+
140
+ def map_inputs_to_backend(
141
+ name: str,
142
+ user_inputs: Dict[str, Any],
143
+ *,
144
+ base_url: str,
145
+ api_key: Optional[str] = None,
146
+ secret_key: Optional[str] = None,
147
+ ) -> Dict[str, Any]:
148
+ """
149
+ Produce the exact payload the backend expects for ``eval_name``, by:
150
+ 1. Looking up the eval's required_keys from the cached registry.
151
+ 2. Taking each required key from user_inputs directly if present.
152
+ 3. Otherwise resolving via known aliases (e.g. ``output`` → ``response``).
153
+ 4. Dropping any keys the backend doesn't accept (the api is strict).
154
+
155
+ If the eval isn't in the registry (unknown name, registry load failed),
156
+ falls back to passing user_inputs through unmodified so the backend
157
+ can return its own validation error.
158
+ """
159
+ required = get_required_keys(
160
+ name, base_url=base_url, api_key=api_key, secret_key=secret_key
161
+ )
162
+ if not required:
163
+ return dict(user_inputs)
164
+
165
+ mapped: Dict[str, Any] = {}
166
+ for key in required:
167
+ if key in user_inputs:
168
+ mapped[key] = user_inputs[key]
169
+ continue
170
+ for alias in _KEY_ALIASES.get(key, ()):
171
+ if alias in user_inputs and alias != key:
172
+ mapped[key] = user_inputs[alias]
173
+ break
174
+ return mapped
175
+
176
+
177
+ def list_known_names(
178
+ *,
179
+ base_url: str,
180
+ api_key: Optional[str] = None,
181
+ secret_key: Optional[str] = None,
182
+ ) -> Set[str]:
183
+ """Names of all evals the backend currently has registered."""
184
+ return set(load_registry(base_url, api_key, secret_key).keys())
@@ -0,0 +1,368 @@
1
+ """
2
+ Engine implementations for evaluate().
3
+
4
+ LocalEngine — wraps BaseMetric subclasses (no API key)
5
+ TuringEngine — wraps the cloud Evaluator HTTP client (template-based evals)
6
+ LLMEngine — wraps CustomLLMJudge + LiteLLMProvider
7
+ """
8
+
9
+ import inspect
10
+ import json as _json
11
+ import time
12
+ from abc import ABC, abstractmethod
13
+ from typing import Any, Dict, Optional
14
+
15
+ from .result import EvalResult
16
+
17
+
18
+ # ---------------------------------------------------------------------------
19
+ # Helpers
20
+ # ---------------------------------------------------------------------------
21
+
22
+ def _try_parse_json(val: Any) -> Any:
23
+ """Try to parse a string as JSON, returning the parsed value or original."""
24
+ if isinstance(val, str):
25
+ try:
26
+ return _json.loads(val)
27
+ except (_json.JSONDecodeError, TypeError):
28
+ return val
29
+ return val
30
+
31
+
32
+ def _normalise_score(output: Any) -> Optional[float]:
33
+ """Convert various metric output types to a 0-1 float."""
34
+ if output is None:
35
+ return None
36
+ if isinstance(output, bool):
37
+ return 1.0 if output else 0.0
38
+ if isinstance(output, (int, float)):
39
+ return float(output)
40
+ if isinstance(output, str):
41
+ try:
42
+ return float(output)
43
+ except ValueError:
44
+ low = output.strip().lower()
45
+ if low in ("true", "yes", "pass", "passed"):
46
+ return 1.0
47
+ if low in ("false", "no", "fail", "failed"):
48
+ return 0.0
49
+ return None
50
+
51
+
52
+ def _extract_result(eval_name: str, batch: Any, latency_ms: float) -> EvalResult:
53
+ """Pull the first result out of a BatchRunResult-like object."""
54
+ r = batch.eval_results[0] if batch.eval_results else None
55
+ if r is None:
56
+ return EvalResult(
57
+ eval_name=eval_name,
58
+ status="failed",
59
+ error="Evaluator returned no results",
60
+ latency_ms=latency_ms,
61
+ )
62
+ return EvalResult(
63
+ eval_name=eval_name,
64
+ score=_normalise_score(r.output),
65
+ reason=r.reason or "",
66
+ latency_ms=latency_ms,
67
+ metadata={
68
+ k: getattr(r, k, None)
69
+ for k in ("eval_id", "output_type")
70
+ if getattr(r, k, None) is not None
71
+ },
72
+ )
73
+
74
+
75
+ # ---------------------------------------------------------------------------
76
+ # Base
77
+ # ---------------------------------------------------------------------------
78
+
79
+ class Engine(ABC):
80
+ """Base class for evaluation engines."""
81
+
82
+ @abstractmethod
83
+ def run(
84
+ self,
85
+ eval_name: str,
86
+ inputs: Dict[str, Any],
87
+ *,
88
+ model: Optional[str] = None,
89
+ prompt: Optional[str] = None,
90
+ config: Optional[Dict[str, Any]] = None,
91
+ ) -> EvalResult:
92
+ ...
93
+
94
+
95
+ # ---------------------------------------------------------------------------
96
+ # LocalEngine
97
+ # ---------------------------------------------------------------------------
98
+
99
+ class LocalEngine(Engine):
100
+ """Runs evaluations locally using BaseMetric subclasses."""
101
+
102
+ @staticmethod
103
+ def _config_keys() -> set:
104
+ from ..types import ConfigPossibleValues
105
+ return set(ConfigPossibleValues.model_fields.keys())
106
+
107
+ def run(
108
+ self,
109
+ eval_name: str,
110
+ inputs: Dict[str, Any],
111
+ *,
112
+ model: Optional[str] = None,
113
+ prompt: Optional[str] = None,
114
+ config: Optional[Dict[str, Any]] = None,
115
+ ) -> EvalResult:
116
+ _ = model, prompt # unused — local metrics don't need these
117
+ from ..local.registry import get_registry
118
+
119
+ config_keys = self._config_keys()
120
+ merged_config = dict(config or {})
121
+ metric_inputs: Dict[str, Any] = {}
122
+ for k, v in inputs.items():
123
+ (merged_config if k in config_keys else metric_inputs)[k] = v
124
+
125
+ metric = get_registry().create(eval_name, merged_config)
126
+ if metric is None:
127
+ return EvalResult(
128
+ eval_name=eval_name,
129
+ status="failed",
130
+ error=f"Local metric '{eval_name}' not found in registry",
131
+ )
132
+
133
+ metric_input = self._build_metric_input(metric, metric_inputs, inputs)
134
+
135
+ start = time.perf_counter()
136
+ try:
137
+ batch = metric.evaluate([metric_input])
138
+ return _extract_result(eval_name, batch, (time.perf_counter() - start) * 1000)
139
+ except Exception as exc:
140
+ return EvalResult(
141
+ eval_name=eval_name,
142
+ status="failed",
143
+ error=str(exc),
144
+ latency_ms=(time.perf_counter() - start) * 1000,
145
+ )
146
+
147
+ @staticmethod
148
+ def _build_metric_input(
149
+ metric: Any,
150
+ inputs: Dict[str, Any],
151
+ all_user_kwargs: Dict[str, Any],
152
+ ) -> Dict[str, Any]:
153
+ """Map user-facing kwargs to the Pydantic input fields the metric expects.
154
+
155
+ Common mappings:
156
+ output -> response
157
+ keyword/context/expected_output -> expected_response
158
+ context (str/list) -> contexts (list)
159
+ input/question -> query
160
+ expected_output/expected_response/ground_truth -> reference
161
+ """
162
+ model_fields = set(metric.input_model.model_fields.keys())
163
+ # Merge both dicts so all user kwargs are searchable
164
+ combined = {**all_user_kwargs, **inputs}
165
+ mapped = {k: v for k, v in combined.items() if k in model_fields}
166
+
167
+ # output -> response
168
+ if "response" not in mapped and "output" in combined:
169
+ if "response" in model_fields:
170
+ mapped["response"] = combined["output"]
171
+
172
+ # expected_response fallback chain (for string metrics)
173
+ if "expected_response" not in mapped and "expected_response" in model_fields:
174
+ for src in ("expected_response", "keyword", "context", "expected_output"):
175
+ if src in combined:
176
+ mapped["expected_response"] = combined[src]
177
+ break
178
+
179
+ # context (str or list) -> contexts (list) — for RAG metrics
180
+ if "contexts" not in mapped and "contexts" in model_fields:
181
+ for src in ("context", "contexts"):
182
+ if src in combined:
183
+ val = combined[src]
184
+ mapped["contexts"] = val if isinstance(val, list) else [val]
185
+ break
186
+
187
+ # input/question -> query — for RAG metrics
188
+ if "query" not in mapped and "query" in model_fields:
189
+ for src in ("query", "input", "question"):
190
+ if src in combined:
191
+ mapped["query"] = combined[src]
192
+ break
193
+
194
+ # expected_output/ground_truth -> reference — for RAG metrics
195
+ if "reference" not in mapped and "reference" in model_fields:
196
+ for src in ("reference", "expected_response", "expected_output", "ground_truth"):
197
+ if src in combined:
198
+ mapped["reference"] = combined[src]
199
+ break
200
+
201
+ # expected -> expected (for structured metrics, also try expected_output)
202
+ if "expected" not in mapped and "expected" in model_fields:
203
+ for src in ("expected", "expected_output"):
204
+ if src in combined:
205
+ mapped["expected"] = _try_parse_json(combined[src])
206
+ break
207
+
208
+ # schema -> schema (try expected_response as schema for structured metrics)
209
+ if "schema" not in mapped and "schema" in model_fields:
210
+ if "expected_response" in combined:
211
+ mapped["schema"] = _try_parse_json(combined["expected_response"])
212
+
213
+ return mapped
214
+
215
+
216
+ # ---------------------------------------------------------------------------
217
+ # TuringEngine
218
+ # ---------------------------------------------------------------------------
219
+
220
+ class TuringEngine(Engine):
221
+ """Runs template-based evaluations on the Turing (FutureAGI) cloud platform.
222
+
223
+ Custom prompts are NOT supported — use ``engine="llm"`` instead.
224
+ """
225
+
226
+ def __init__(
227
+ self,
228
+ fi_api_key: Optional[str] = None,
229
+ fi_secret_key: Optional[str] = None,
230
+ fi_base_url: Optional[str] = None,
231
+ ):
232
+ self._api_key = fi_api_key
233
+ self._secret_key = fi_secret_key
234
+ self._base_url = fi_base_url
235
+ self._evaluator: Optional[Any] = None
236
+
237
+ def _get_evaluator(self):
238
+ if self._evaluator is None:
239
+ from ..evaluator import Evaluator
240
+ self._evaluator = Evaluator(
241
+ fi_api_key=self._api_key,
242
+ fi_secret_key=self._secret_key,
243
+ fi_base_url=self._base_url,
244
+ )
245
+ return self._evaluator
246
+
247
+ def run(
248
+ self,
249
+ eval_name: str,
250
+ inputs: Dict[str, Any],
251
+ *,
252
+ model: Optional[str] = None,
253
+ prompt: Optional[str] = None,
254
+ config: Optional[Dict[str, Any]] = None,
255
+ ) -> EvalResult:
256
+ _ = config # unused — template config comes from the template class
257
+ if prompt:
258
+ return EvalResult(
259
+ eval_name=eval_name or "custom_prompt",
260
+ status="failed",
261
+ error=(
262
+ "Custom prompts are not supported on the Turing engine. "
263
+ "Use engine='llm' with a model like 'claude-4.5-sonnet' "
264
+ "or 'gpt-5' instead."
265
+ ),
266
+ )
267
+
268
+ # Dynamic registry drives input filtering — backend is source of truth
269
+ # for required_keys. Falls back to pass-through when eval isn't known
270
+ # (registry miss, offline, etc.) so the backend can return its own
271
+ # validation error.
272
+ from .cloud_registry import map_inputs_to_backend
273
+
274
+ evaluator = self._get_evaluator()
275
+ mapped_inputs = map_inputs_to_backend(
276
+ eval_name,
277
+ inputs,
278
+ base_url=evaluator._base_url,
279
+ api_key=getattr(evaluator, "_fi_api_key", None) or self._api_key,
280
+ secret_key=getattr(evaluator, "_fi_secret_key", None) or self._secret_key,
281
+ )
282
+
283
+ # Template class is optional — if the SDK hasn't shipped a class
284
+ # for a new backend eval, just send the name as a string.
285
+ template_cls = self._resolve_template_class(eval_name)
286
+ eval_arg: Any = template_cls() if template_cls is not None else eval_name
287
+
288
+ start = time.perf_counter()
289
+ try:
290
+ batch = evaluator.evaluate(
291
+ eval_templates=eval_arg,
292
+ inputs=mapped_inputs,
293
+ model_name=model,
294
+ )
295
+ return _extract_result(eval_name, batch, (time.perf_counter() - start) * 1000)
296
+ except Exception as exc:
297
+ return EvalResult(
298
+ eval_name=eval_name,
299
+ status="failed",
300
+ error=str(exc),
301
+ latency_ms=(time.perf_counter() - start) * 1000,
302
+ )
303
+
304
+ @staticmethod
305
+ def _resolve_template_class(eval_name: str) -> Optional[type]:
306
+ """Find the EvalTemplate subclass for a given eval_name."""
307
+ from ..templates import EvalTemplate
308
+ from .. import templates as tmpl_mod
309
+
310
+ return next(
311
+ (
312
+ obj
313
+ for _, obj in inspect.getmembers(tmpl_mod, inspect.isclass)
314
+ if issubclass(obj, EvalTemplate)
315
+ and obj is not EvalTemplate
316
+ and getattr(obj, "eval_name", None) == eval_name
317
+ ),
318
+ None,
319
+ )
320
+
321
+
322
+ # ---------------------------------------------------------------------------
323
+ # LLMEngine
324
+ # ---------------------------------------------------------------------------
325
+
326
+ class LLMEngine(Engine):
327
+ """Runs evaluations using an LLM as a judge (via LiteLLM)."""
328
+
329
+ def run(
330
+ self,
331
+ eval_name: str,
332
+ inputs: Dict[str, Any],
333
+ *,
334
+ model: Optional[str] = None,
335
+ prompt: Optional[str] = None,
336
+ config: Optional[Dict[str, Any]] = None,
337
+ ) -> EvalResult:
338
+ if not prompt:
339
+ return EvalResult(
340
+ eval_name=eval_name,
341
+ status="failed",
342
+ error="LLMEngine requires a 'prompt' parameter",
343
+ )
344
+
345
+ from ..llm.providers.litellm import LiteLLMProvider
346
+ from ..metrics.llm_as_judges.custom_judge.metric import CustomLLMJudge
347
+
348
+ judge = CustomLLMJudge(
349
+ provider=LiteLLMProvider(),
350
+ config={
351
+ "grading_criteria": prompt,
352
+ "model": model or "gpt-4o",
353
+ **(config or {}),
354
+ },
355
+ )
356
+
357
+ name = eval_name or "custom_llm_judge"
358
+ start = time.perf_counter()
359
+ try:
360
+ batch = judge.evaluate([inputs])
361
+ return _extract_result(name, batch, (time.perf_counter() - start) * 1000)
362
+ except Exception as exc:
363
+ return EvalResult(
364
+ eval_name=name,
365
+ status="failed",
366
+ error=str(exc),
367
+ latency_ms=(time.perf_counter() - start) * 1000,
368
+ )