agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/evals/evaluator.py ADDED
@@ -0,0 +1,721 @@
1
+ import inspect
2
+ import json
3
+ import logging
4
+ import os
5
+ from concurrent.futures import ThreadPoolExecutor, as_completed, TimeoutError
6
+ from typing import Any, Dict, List, Optional, Union
7
+
8
+ from requests import Response
9
+
10
+ from fi.api.auth import APIKeyAuth, ResponseHandler
11
+ from fi.api.types import HttpMethod, RequestConfig
12
+ from fi.evals.execution import Execution, _normalize_status
13
+ from fi.evals.templates import EvalTemplate
14
+ from fi.evals.types import BatchRunResult, EvalResult
15
+ from fi.utils.errors import InvalidAuthError
16
+ from fi.utils.routes import Routes
17
+
18
+ def _coerce_to_api_input(value: Any) -> Any:
19
+ """Serialize rich native Python objects into API-supported input values."""
20
+ if isinstance(value, dict):
21
+ return json.dumps(value)
22
+ if isinstance(value, list):
23
+ if all(isinstance(v, str) for v in value):
24
+ return value
25
+ if all(
26
+ isinstance(v, list) and all(isinstance(x, str) for x in v)
27
+ for v in value
28
+ ):
29
+ return value
30
+ return json.dumps(value)
31
+ return value
32
+
33
+
34
+ class EvalResponseHandler(ResponseHandler[BatchRunResult, None]):
35
+ """Handles responses for evaluation requests"""
36
+
37
+ @classmethod
38
+ def _parse_success(cls, response: Response) -> BatchRunResult:
39
+ return cls.convert_to_batch_results(response.json())
40
+
41
+ @classmethod
42
+ def _handle_error(cls, response: Response) -> None:
43
+ if response.status_code == 400:
44
+ raise Exception(
45
+ f"Evaluation failed with a 400 Bad Request. Please check your input data and evaluation configuration. Response: {response.text}"
46
+ )
47
+ elif response.status_code == 403:
48
+ raise InvalidAuthError()
49
+ else:
50
+ raise Exception(
51
+ f"Error in evaluation: {response.status_code}, response: {response.text}"
52
+ )
53
+
54
+ @classmethod
55
+ def convert_to_batch_results(cls, response: Dict[str, Any]) -> BatchRunResult:
56
+ """
57
+ Convert API response to BatchRunResult
58
+
59
+ Args:
60
+ response: Raw API response dictionary
61
+
62
+ Returns:
63
+ BatchRunResult containing evaluation results
64
+ """
65
+ eval_results = []
66
+
67
+ # The revamped backend (post 2026-04-12) returns pure snake_case:
68
+ # {"result": [{"evaluations": [
69
+ # {"name", "reason", "runtime", "output", "output_type",
70
+ # "eval_id", "model"?, "error_localizer_enabled"?,
71
+ # "error_localizer"?}
72
+ # ]}]}
73
+ # Async / error-localization paths may return the eval wrapped in
74
+ # {"eval_status": "...", "result": <eval>} — handle that too.
75
+ for result in response.get("result", []) or []:
76
+ if isinstance(result, dict) and "evaluations" in result:
77
+ entries = result.get("evaluations", []) or []
78
+ else:
79
+ entries = [result] if isinstance(result, dict) else []
80
+
81
+ for evaluation in entries:
82
+ if not isinstance(evaluation, dict):
83
+ continue
84
+ eval_results.append(
85
+ EvalResult(
86
+ name=evaluation.get("name", ""),
87
+ output=evaluation.get("output", evaluation.get("value")),
88
+ reason=evaluation.get("reason", ""),
89
+ runtime=evaluation.get("runtime", 0),
90
+ output_type=evaluation.get("output_type", ""),
91
+ eval_id=evaluation.get("eval_id", ""),
92
+ model=evaluation.get("model"),
93
+ error_localizer_enabled=evaluation.get(
94
+ "error_localizer_enabled"
95
+ ),
96
+ error_localizer=evaluation.get("error_localizer"),
97
+ )
98
+ )
99
+
100
+ return BatchRunResult(eval_results=eval_results)
101
+
102
+
103
+ class EvalInfoResponseHandler(ResponseHandler[dict, None]):
104
+ """Handles responses for evaluation info requests"""
105
+
106
+ @classmethod
107
+ def _parse_success(cls, response: Response) -> dict:
108
+ data = response.json()
109
+ if "result" in data:
110
+ return data["result"]
111
+ else:
112
+ raise Exception(f"Failed to get evaluation info: {data}")
113
+
114
+ @classmethod
115
+ def _handle_error(cls, response: Response) -> None:
116
+ if response.status_code == 400:
117
+ response.raise_for_status()
118
+ if response.status_code == 403:
119
+ raise InvalidAuthError()
120
+ raise Exception(f"Failed to get evaluation info: {response.status_code}")
121
+
122
+
123
+ class Evaluator(APIKeyAuth):
124
+ """Client for evaluating LLM test cases"""
125
+
126
+ def __init__(
127
+ self,
128
+ fi_api_key: Optional[str] = None,
129
+ fi_secret_key: Optional[str] = None,
130
+ fi_base_url: Optional[str] = None,
131
+ **kwargs,
132
+ ) -> None:
133
+ """
134
+ Initialize the Eval Client
135
+
136
+ Args:
137
+ fi_api_key: API key
138
+ fi_secret_key: Secret key
139
+ fi_base_url: Base URL
140
+
141
+ Keyword Args:
142
+ timeout: Optional timeout value in seconds (default: 200)
143
+ max_queue_bound: Optional maximum queue size (default: 5000)
144
+ max_workers: Optional maximum number of workers (default: 8)
145
+ langfuse_secret_key: Optional Langfuse secret key
146
+ langfuse_public_key: Optional Langfuse public key
147
+ langfuse_host: Optional Langfuse host
148
+ """
149
+ super().__init__(fi_api_key, fi_secret_key, fi_base_url, **kwargs)
150
+ self._max_workers = kwargs.get("max_workers", 8) # Default to 8 if not provided
151
+
152
+ # Handle Langfuse credentials
153
+ self.langfuse_secret_key = kwargs.get("langfuse_secret_key") or os.getenv("LANGFUSE_SECRET_KEY")
154
+ self.langfuse_public_key = kwargs.get("langfuse_public_key") or os.getenv("LANGFUSE_PUBLIC_KEY")
155
+ self.langfuse_host = kwargs.get("langfuse_host") or os.getenv("LANGFUSE_HOST")
156
+
157
+ # Instance-level cache for _get_eval_info results.
158
+ # Previously decorated with @lru_cache, but @lru_cache on a bound
159
+ # method (one taking `self`) holds a strong reference to the
160
+ # instance via the cache key, preventing the Evaluator from being
161
+ # garbage-collected. In long-running processes (Celery / Temporal
162
+ # workers, FastAPI lifespans) that creates a slow memory leak.
163
+ self._eval_info_cache: Dict[str, Dict[str, Any]] = {}
164
+
165
+
166
+ def evaluate(
167
+ self,
168
+ eval_templates: Union[str, type[EvalTemplate]],
169
+ inputs: Dict[str, Any],
170
+ timeout: Optional[int] = None,
171
+ model_name: Optional[str] = None,
172
+ custom_eval_name: Optional[str] = None,
173
+ trace_eval: Optional[bool] = False,
174
+ platform: Optional[str] = None,
175
+ is_async: Optional[bool] = False,
176
+ error_localizer: Optional[bool] = False,
177
+ eval_config: Optional[Dict[str, Any]] = None,
178
+ **kwargs,
179
+ ) -> BatchRunResult:
180
+ """
181
+ Run a single or batch of evaluations independently
182
+
183
+ Args:
184
+ eval_templates: Evaluation name string (e.g., "Factual Accuracy")
185
+ inputs: Single test case or list of test cases
186
+ timeout: Optional timeout value for the evaluation
187
+ model_name: Optional model name to use for the evaluation for Future AGI Agents
188
+ span_id: Optional span_id to attach to the evaluation. If not provided, it will be retrieved from the OpenTelemetry context if available.
189
+ custom_eval_name: Optional custom evaluation name to use for the evaluation. If not provided, eval will not be added to the span.
190
+ Returns:
191
+ BatchRunResult containing evaluation results
192
+
193
+ Raises:
194
+ ValidationError: If the inputs do not match the evaluation templates
195
+ Exception: If the API request fails
196
+ """
197
+ if platform:
198
+ if isinstance(eval_templates, str) and isinstance(inputs, dict) and custom_eval_name:
199
+ return self._configure_evaluations(
200
+ eval_templates=eval_templates,
201
+ inputs=inputs,
202
+ platform=platform,
203
+ custom_eval_name=custom_eval_name,
204
+ model_name=model_name,
205
+ **kwargs
206
+ )
207
+ else:
208
+ raise ValueError("Invalid arguments for platform configuration")
209
+
210
+
211
+ def _extract_name(t) -> str | None:
212
+ if isinstance(t, str):
213
+ return t
214
+ if isinstance(t, EvalTemplate):
215
+ return t.eval_name
216
+ if inspect.isclass(t) and issubclass(t, EvalTemplate):
217
+ return t.eval_name
218
+ return None
219
+
220
+ eval_name = _extract_name(
221
+ eval_templates[0] if isinstance(eval_templates, list) else eval_templates
222
+ )
223
+
224
+ span_id = None
225
+ project_name = None
226
+ if trace_eval:
227
+ if not custom_eval_name:
228
+ trace_eval = False
229
+ logging.warning("Failed to trace the evaluation. Please set the custom_eval_name.")
230
+ else:
231
+ try:
232
+ from opentelemetry import trace
233
+
234
+ current_span = trace.get_current_span()
235
+ if current_span and current_span.is_recording():
236
+ span_context = current_span.get_span_context()
237
+ if span_context.is_valid:
238
+ span_id = format(span_context.span_id, "016x")
239
+ tracer_provider = trace.get_tracer_provider()
240
+ if hasattr(tracer_provider, "resource"):
241
+ attributes = tracer_provider.resource.attributes
242
+ project_name = attributes.get("project_name")
243
+
244
+ if not project_name:
245
+ trace_eval = False
246
+ logging.warning(
247
+ "Could not determine project_name from OpenTelemetry context. "
248
+ "Skipping trace_eval for this evaluation."
249
+ )
250
+
251
+ except ImportError:
252
+ logging.exception(
253
+ "Future AGI SDK not found. "
254
+ "Please install 'fi-instrumentation-otel' to automatically enrich the evaluation with project context."
255
+ )
256
+ return
257
+
258
+ if eval_name is None:
259
+ raise TypeError(
260
+ "Unsupported eval_templates argument. "
261
+ "Expect eval template class/obj or name str."
262
+ )
263
+
264
+ # Dynamic registry: filter user-supplied inputs to only the keys the
265
+ # backend currently accepts for this eval. The api rejects supersets
266
+ # (e.g. {output,input,context} for a template that only wants
267
+ # {output}), so this can't be a pass-through. If the registry fetch
268
+ # fails or the name is unknown, leave inputs untouched.
269
+ if kwargs.get("skip_input_mapping") is not True and isinstance(inputs, dict):
270
+ try:
271
+ from fi.evals.core.cloud_registry import map_inputs_to_backend
272
+ inputs = map_inputs_to_backend(
273
+ eval_name,
274
+ inputs,
275
+ base_url=self._base_url,
276
+ api_key=self._fi_api_key,
277
+ secret_key=self._fi_secret_key,
278
+ )
279
+ except Exception as exc:
280
+ logging.debug("Dynamic input mapping skipped: %s", exc)
281
+
282
+ # The api validator accepts only strings, list[str], or list[list[str]].
283
+ # JSON-serialize dict / list-of-dicts values (e.g. conversation messages)
284
+ # so users can pass native Python objects without manually stringifying.
285
+ if isinstance(inputs, dict):
286
+ inputs = {k: _coerce_to_api_input(v) for k, v in inputs.items()}
287
+
288
+ final_api_payload = {
289
+ "eval_name": eval_name,
290
+ "inputs": inputs,
291
+ "model": model_name,
292
+ "span_id": span_id,
293
+ "custom_eval_name": custom_eval_name,
294
+ "trace_eval": trace_eval,
295
+ "is_async": is_async,
296
+ "error_localizer": error_localizer,
297
+ }
298
+
299
+ if eval_config:
300
+ final_api_payload["config"] = {"params": eval_config}
301
+
302
+
303
+ all_results = []
304
+ failed_inputs = []
305
+ with ThreadPoolExecutor(max_workers=self._max_workers) as executor:
306
+ # Submit the batch only once
307
+ future = executor.submit(
308
+ self.request,
309
+ config=RequestConfig(
310
+ method=HttpMethod.POST,
311
+ url=f"{self._base_url}/{Routes.evaluatev2.value}",
312
+ json=final_api_payload,
313
+ timeout=timeout or self._default_timeout,
314
+ ),
315
+ response_handler=EvalResponseHandler,
316
+ )
317
+ future_to_input = {future: inputs} # map single future to all inputs
318
+
319
+ for future in as_completed(future_to_input):
320
+ try:
321
+ response: BatchRunResult = future.result(timeout=timeout or self._default_timeout)
322
+ all_results.extend(response.eval_results)
323
+ except TimeoutError:
324
+ input_case = future_to_input[future]
325
+ logging.error(f"Evaluation timed out for input: {input_case}")
326
+ failed_inputs.append(input_case)
327
+ all_results.append(
328
+ EvalResult(
329
+ name=eval_name,
330
+ output=None,
331
+ reason=f"Evaluation timed out after {timeout or self._default_timeout}s",
332
+ runtime=0,
333
+ )
334
+ )
335
+ except Exception as exc:
336
+ input_case = future_to_input[future]
337
+ logging.error(f"Evaluation failed for input {input_case}: {str(exc)}")
338
+ failed_inputs.append(input_case)
339
+ all_results.append(
340
+ EvalResult(
341
+ name=eval_name,
342
+ output=None,
343
+ reason=str(exc),
344
+ runtime=0,
345
+ )
346
+ )
347
+
348
+ if failed_inputs:
349
+ logging.warning(f"Failed to evaluate {len(failed_inputs)} inputs out of {len(inputs)} total inputs")
350
+
351
+ # Automatically enrich current span with evaluation results
352
+ result = BatchRunResult(eval_results=all_results)
353
+ try:
354
+ from fi.evals.otel.enrichment import enrich_span_with_batch_result, is_auto_enrichment_enabled
355
+ if is_auto_enrichment_enabled():
356
+ enriched_count = enrich_span_with_batch_result(result)
357
+ if enriched_count > 0:
358
+ logging.debug(f"Enriched active span with {enriched_count} evaluation results")
359
+ except ImportError:
360
+ pass # OTEL enrichment not available
361
+ except Exception as e:
362
+ logging.debug(f"Failed to enrich span with evaluation results: {e}")
363
+
364
+ return result
365
+
366
+
367
+ def get_eval_result(self, eval_id: str):
368
+ """
369
+ Get the raw evaluation status payload by ID (unparsed).
370
+
371
+ For a higher-level handle that understands the status envelope
372
+ and can be awaited, see :py:meth:`get_execution`.
373
+ """
374
+ url = f"{self._base_url}/{Routes.get_eval_result.value}"
375
+ response = self.request(
376
+ config=RequestConfig(
377
+ method=HttpMethod.GET,
378
+ url=url,
379
+ params={"eval_id": eval_id},
380
+ timeout=self._default_timeout,
381
+ ),
382
+ )
383
+
384
+ return response.json()
385
+
386
+ # ------------------------------------------------------------------
387
+ # Async submission / execution handles
388
+ # ------------------------------------------------------------------
389
+
390
+ def submit(
391
+ self,
392
+ eval_templates: Union[str, type[EvalTemplate]],
393
+ inputs: Dict[str, Any],
394
+ *,
395
+ model_name: Optional[str] = None,
396
+ custom_eval_name: Optional[str] = None,
397
+ trace_eval: bool = False,
398
+ error_localizer: bool = False,
399
+ timeout: Optional[int] = None,
400
+ **kwargs: Any,
401
+ ) -> Execution:
402
+ """
403
+ Submit an eval for async execution and return an :class:`Execution`
404
+ handle immediately. Use ``handle.wait()`` or
405
+ :py:meth:`get_execution` to poll for completion.
406
+
407
+ This is the non-blocking equivalent of :py:meth:`evaluate` —
408
+ internally it always sets ``is_async=True`` so the backend records
409
+ the evaluation and starts a worker without holding the HTTP
410
+ connection open.
411
+ """
412
+
413
+ def _extract_name(t: Any) -> Optional[str]:
414
+ if isinstance(t, str):
415
+ return t
416
+ if isinstance(t, EvalTemplate):
417
+ return t.eval_name
418
+ if inspect.isclass(t) and issubclass(t, EvalTemplate):
419
+ return t.eval_name
420
+ return None
421
+
422
+ eval_name = _extract_name(
423
+ eval_templates[0] if isinstance(eval_templates, list) else eval_templates
424
+ )
425
+ if eval_name is None:
426
+ raise TypeError(
427
+ "Unsupported eval_templates argument. "
428
+ "Expect eval template class/obj or name str."
429
+ )
430
+
431
+ payload = {
432
+ "eval_name": eval_name,
433
+ "inputs": inputs,
434
+ "model": model_name,
435
+ "span_id": kwargs.get("span_id"),
436
+ "custom_eval_name": custom_eval_name,
437
+ "trace_eval": trace_eval,
438
+ "is_async": True,
439
+ "error_localizer": error_localizer,
440
+ }
441
+
442
+ response = self.request(
443
+ config=RequestConfig(
444
+ method=HttpMethod.POST,
445
+ url=f"{self._base_url}/{Routes.evaluatev2.value}",
446
+ json=payload,
447
+ timeout=timeout or self._default_timeout,
448
+ ),
449
+ )
450
+ body = response.json() if hasattr(response, "json") else {}
451
+
452
+ # Backend responds with:
453
+ # {"status": true, "result": [{"evaluations": [{name, output_type, eval_id}]}]}
454
+ # on success, or
455
+ # {"status": false, "result": {...error dict...}}
456
+ # on validation failure.
457
+ if not body.get("status", True):
458
+ raise RuntimeError(
459
+ f"Async submit rejected by backend: {body.get('result')}"
460
+ )
461
+
462
+ results = body.get("result") or []
463
+ if not isinstance(results, list) or not results:
464
+ raise RuntimeError(
465
+ f"Async submit did not return a result list (response: {body})"
466
+ )
467
+ evaluations = results[0].get("evaluations") or []
468
+ first_eval = evaluations[0] if evaluations else {}
469
+ execution_id = first_eval.get("eval_id")
470
+ if not execution_id:
471
+ raise RuntimeError(
472
+ f"Async submit did not return an eval_id (response: {body})"
473
+ )
474
+
475
+ handle = Execution(
476
+ id=str(execution_id),
477
+ kind="eval",
478
+ status="pending",
479
+ )
480
+ handle._refresher = lambda eid=execution_id: self._refresh_eval_execution(eid)
481
+ return handle
482
+
483
+ def get_execution(self, execution_id: str) -> Execution:
484
+ """
485
+ Fetch the latest state of an async single-eval execution by ID.
486
+
487
+ Returns an :class:`Execution` handle with an attached refresher
488
+ closure — call ``handle.wait()`` to block until completion.
489
+ """
490
+ handle = self._refresh_eval_execution(execution_id)
491
+ handle._refresher = (
492
+ lambda eid=execution_id: self._refresh_eval_execution(eid)
493
+ )
494
+ return handle
495
+
496
+ def _refresh_eval_execution(self, execution_id: str) -> Execution:
497
+ url = f"{self._base_url}/{Routes.get_eval_result.value}"
498
+ response = self.request(
499
+ config=RequestConfig(
500
+ method=HttpMethod.GET,
501
+ url=url,
502
+ params={"eval_id": execution_id},
503
+ timeout=self._default_timeout,
504
+ ),
505
+ )
506
+ body = response.json() if hasattr(response, "json") else {}
507
+ # Envelope: {"status": true, "result": {"eval_status": ..., "result": <body|str>}}
508
+ payload = body.get("result") or {}
509
+ status = _normalize_status(payload.get("eval_status"))
510
+ raw_result = payload.get("result")
511
+ error_message = payload.get("error_message")
512
+
513
+ parsed_result: Any = None
514
+ error_localizer: Optional[Dict[str, Any]] = None
515
+ if isinstance(raw_result, dict):
516
+ # Completed / failed state — full eval record.
517
+ parsed_result = EvalResult(
518
+ name=raw_result.get("name", ""),
519
+ output=raw_result.get("output", raw_result.get("value")),
520
+ reason=raw_result.get("reason"),
521
+ runtime=raw_result.get("runtime", 0),
522
+ output_type=raw_result.get("output_type"),
523
+ eval_id=str(raw_result.get("eval_id", execution_id)),
524
+ model=raw_result.get("model"),
525
+ error_localizer_enabled=raw_result.get("error_localizer_enabled"),
526
+ error_localizer=raw_result.get("error_localizer"),
527
+ )
528
+ error_localizer = raw_result.get("error_localizer")
529
+ error_message = raw_result.get("error_message") or error_message
530
+ # else: raw_result is a human-readable string like "Evaluation is
531
+ # being processed." — leave parsed_result as None.
532
+
533
+ return Execution(
534
+ id=str(execution_id),
535
+ kind="eval",
536
+ status=status,
537
+ result=parsed_result,
538
+ error_message=error_message,
539
+ error_localizer=error_localizer,
540
+ )
541
+
542
+
543
+ def _configure_evaluations(
544
+ self,
545
+ eval_templates: str,
546
+ inputs: Dict[str, Any],
547
+ platform: str,
548
+ custom_eval_name: str,
549
+ model_name: Optional[str] = None,
550
+ **kwargs,
551
+ ) -> Dict[str, Any]:
552
+ """
553
+ Configure evaluations on a specified platform.
554
+
555
+ This will not return any evaluation results, but rather a
556
+ confirmation message from the backend.
557
+
558
+ Args:
559
+ eval_config: The evaluation configuration dictionary.
560
+ platform: The platform to which the evaluations should be sent.
561
+ timeout: Optional timeout for the API request.
562
+ **kwargs: Additional configuration parameters to be sent with the request.
563
+
564
+ Returns:
565
+ A dictionary containing the backend's response message.
566
+ """
567
+ try:
568
+ from fi.evals.otel_utils import _get_current_otel_span
569
+
570
+ if platform == "langfuse":
571
+ kwargs["langfuse_secret_key"] = self.langfuse_secret_key
572
+ kwargs["langfuse_public_key"] = self.langfuse_public_key
573
+ kwargs["langfuse_host"] = self.langfuse_host
574
+
575
+ current_span = _get_current_otel_span()
576
+ if current_span:
577
+ span_context = current_span.get_span_context()
578
+ if span_context.is_valid:
579
+ span_id = format(span_context.span_id, "016x")
580
+ trace_id = format(span_context.trace_id, "032x")
581
+ kwargs["span_id"] = span_id
582
+ kwargs["trace_id"] = trace_id
583
+
584
+ # Check if span_id and trace_id are present in kwargs
585
+ if "span_id" not in kwargs or "trace_id" not in kwargs:
586
+ logging.warning(
587
+ "span_id and/or trace_id not found in kwargs ."
588
+ "Please run this function within a span context."
589
+ )
590
+ return
591
+
592
+ api_payload = {
593
+ "eval_config": {
594
+ "eval_templates": eval_templates,
595
+ "inputs": inputs,
596
+ "model_name": model_name
597
+ },
598
+ "custom_eval_name": custom_eval_name,
599
+ "platform": platform,
600
+ **kwargs,
601
+ }
602
+
603
+ response = self.request(
604
+ config=RequestConfig(
605
+ method=HttpMethod.POST,
606
+ url=f"{self._base_url}/{Routes.configure_evaluations.value}",
607
+ json=api_payload,
608
+ timeout=self._default_timeout,
609
+ ),
610
+ )
611
+
612
+ if response.status_code != 200:
613
+ logging.warning(
614
+ f"Received non-200 status code from backend: {response.status_code}. "
615
+ f"Response: {response.text}"
616
+ )
617
+
618
+ return response.json()
619
+
620
+ except ImportError:
621
+ logging.exception(
622
+ "Future AGI SDK not found. "
623
+ "Please install 'fi-instrumentation-otel' to use these evaluations."
624
+ )
625
+ return
626
+
627
+
628
+ def _get_eval_info(self, eval_name: str) -> Dict[str, Any]:
629
+ cached = self._eval_info_cache.get(eval_name)
630
+ if cached is not None:
631
+ return cached
632
+
633
+ url = (
634
+ self._base_url
635
+ + "/"
636
+ + Routes.get_eval_templates.value
637
+ )
638
+ response = self.request(
639
+ config=RequestConfig(method=HttpMethod.GET, url=url),
640
+ response_handler=EvalInfoResponseHandler,
641
+ )
642
+ eval_info = next((item for item in response if item["name"] == eval_name), None)
643
+ if eval_info is None:
644
+ raise KeyError(f"Evaluation template '{eval_name}' not found in registry")
645
+ if not eval_info:
646
+ raise Exception(f"Evaluation template with name '{eval_name}' not found")
647
+ self._eval_info_cache[eval_name] = eval_info
648
+ return eval_info
649
+
650
+ def list_evaluations(self):
651
+ """
652
+ Fetch information about all available evaluation templates by getting eval_info
653
+ for each template class defined in templates.py.
654
+
655
+ Returns:
656
+ List[Dict[str, Any]]: List of evaluation template information dictionaries
657
+ """
658
+ config = RequestConfig(method=HttpMethod.GET,
659
+ url=f"{self._base_url}/{Routes.get_eval_templates.value}")
660
+
661
+ response = self.request(config=config, response_handler=EvalInfoResponseHandler)
662
+
663
+ return response
664
+
665
+
666
+ def evaluate_pipeline(
667
+ self,
668
+ project_name: str,
669
+ version : str,
670
+ eval_data : List[Dict[str, Any]],
671
+ ):
672
+ api_payload = {
673
+ "project_name": project_name,
674
+ "version": version,
675
+ "eval_data": eval_data
676
+ }
677
+
678
+ response = self.request(
679
+ config=RequestConfig(
680
+ method=HttpMethod.POST,
681
+ url=f"{self._base_url}/{Routes.evaluate_pipeline.value}",
682
+ json=api_payload,
683
+ timeout=self._default_timeout,
684
+ ),
685
+ )
686
+
687
+ return response.json()
688
+
689
+
690
+ def get_pipeline_results(
691
+ self,
692
+ project_name: str,
693
+ versions : List[str],
694
+ ):
695
+
696
+ if not isinstance(versions, list) or not all(isinstance(v, str) for v in versions):
697
+ raise TypeError("versions must be a list of strings")
698
+
699
+ api_payload = {
700
+ "project_name": project_name,
701
+ "versions": ",".join(versions),
702
+ }
703
+
704
+ response = self.request(
705
+ config=RequestConfig(
706
+ method=HttpMethod.GET,
707
+ url=f"{self._base_url}/{Routes.evaluate_pipeline.value}",
708
+ params=api_payload,
709
+ timeout=self._default_timeout,
710
+ ),
711
+ )
712
+
713
+ return response.json()
714
+
715
+
716
+ # Top-level convenience for the common "list everything" case.
717
+ # The main ``evaluate()`` entrypoint is imported from ``fi.evals.core``.
718
+ def list_evaluations():
719
+ return Evaluator().list_evaluations()
720
+
721
+