agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,267 @@
1
+ """
2
+ Evaluation Framework - Scalable evaluation infrastructure.
3
+
4
+ This module provides a unified framework for running evaluations in different modes:
5
+ - Blocking (synchronous): For development and when results are needed immediately
6
+ - Non-blocking (async): For production with zero latency impact
7
+ - Distributed: For batch processing at scale via pluggable backends
8
+
9
+ Key Features:
10
+ - Trace context propagation across threads/processes
11
+ - Automatic span enrichment with evaluation results
12
+ - Pluggable backends (thread pool built-in, Temporal/Celery/Ray extensible)
13
+ - Write evaluation logic once, run anywhere
14
+
15
+ Quick Start:
16
+ from fi.evals.framework import Evaluator, ExecutionMode, register_evaluation
17
+
18
+ # Define an evaluation
19
+ @register_evaluation
20
+ class MyEval:
21
+ name = "my_eval"
22
+ version = "1.0.0"
23
+
24
+ def evaluate(self, inputs):
25
+ score = compute_score(inputs["response"])
26
+ return {"score": score, "passed": score > 0.7}
27
+
28
+ def get_span_attributes(self, result):
29
+ return {"score": result["score"], "passed": result["passed"]}
30
+
31
+ # Run blocking (development)
32
+ evaluator = Evaluator([MyEval()], mode=ExecutionMode.BLOCKING)
33
+ result = evaluator.run({"response": "..."})
34
+ print(result.results[0].value)
35
+
36
+ # Run non-blocking (production - zero latency)
37
+ evaluator = Evaluator([MyEval()], mode=ExecutionMode.NON_BLOCKING)
38
+ result = evaluator.run({"response": "..."}) # Returns immediately
39
+ batch = result.wait() # Get results when needed
40
+
41
+ Factory Functions:
42
+ # Convenience functions for common configurations
43
+ from fi.evals.framework import blocking_evaluator, async_evaluator
44
+
45
+ # Blocking
46
+ evaluator = blocking_evaluator(MyEval())
47
+ result = evaluator.run(inputs)
48
+
49
+ # Async (non-blocking)
50
+ evaluator = async_evaluator(MyEval(), max_workers=4)
51
+ result = evaluator.run(inputs)
52
+ batch = result.wait()
53
+
54
+ Example with OpenTelemetry:
55
+ from fi.evals.framework import async_evaluator, register_current_span
56
+
57
+ evaluator = async_evaluator(ToxicityEval(), BiasEval())
58
+
59
+ with tracer.start_as_current_span("llm_call") as span:
60
+ register_current_span() # Enable cross-thread enrichment
61
+ response = llm.complete(prompt)
62
+ evaluator.run({"response": response}) # Zero latency
63
+ return response
64
+
65
+ # Span automatically enriched with:
66
+ # eval.toxicity.score, eval.toxicity.status
67
+ # eval.bias.score, eval.bias.status
68
+ """
69
+
70
+ __version__ = "1.0.0"
71
+
72
+ # Core types
73
+ from .types import (
74
+ ExecutionMode,
75
+ EvalStatus,
76
+ FrameworkEvalResult,
77
+ BatchEvalResult,
78
+ EvalInputs,
79
+ SpanAttributes,
80
+ )
81
+
82
+ # Context
83
+ from .context import (
84
+ EvalContext,
85
+ get_current_context,
86
+ create_standalone_context,
87
+ )
88
+
89
+ # Protocols
90
+ from .protocols import (
91
+ BaseEvaluation,
92
+ EvalRegistry,
93
+ register_evaluation,
94
+ create_evaluation,
95
+ )
96
+
97
+ # Enrichment
98
+ from .enrichment import (
99
+ enrich_current_span,
100
+ enrich_span,
101
+ add_eval_event,
102
+ get_current_span,
103
+ is_span_recording,
104
+ flatten_attributes,
105
+ SpanEnricher,
106
+ )
107
+
108
+ # Registry
109
+ from .registry import (
110
+ SpanRegistry,
111
+ register_span,
112
+ get_span,
113
+ unregister_span,
114
+ get_registry,
115
+ register_current_span,
116
+ )
117
+
118
+ # Propagation
119
+ from .propagation import (
120
+ SpanContextPropagator,
121
+ enrich_span_by_context,
122
+ enrich_span_by_ids,
123
+ add_event_by_context,
124
+ ContextCarrier,
125
+ propagate_context,
126
+ propagate_context_lazy,
127
+ )
128
+
129
+ # Evaluators
130
+ from .evaluators import (
131
+ BlockingEvaluator,
132
+ blocking_evaluate,
133
+ NonBlockingEvaluator,
134
+ non_blocking_evaluate,
135
+ EvalFuture,
136
+ BatchEvalFuture,
137
+ EvalResultAggregator,
138
+ )
139
+
140
+ # Backends
141
+ from .backends import (
142
+ Backend,
143
+ BackendConfig,
144
+ TaskHandle,
145
+ TaskStatus,
146
+ ThreadPoolBackend,
147
+ ThreadPoolConfig,
148
+ )
149
+
150
+ # Resilience
151
+ from .resilience import (
152
+ ResilientBackend,
153
+ ResilienceConfig,
154
+ CircuitBreakerConfig,
155
+ RateLimitConfig,
156
+ RetryConfig,
157
+ DegradationConfig,
158
+ HealthCheckConfig,
159
+ wrap_backend,
160
+ )
161
+
162
+ # Built-in evals + builder
163
+ from .evals import (
164
+ custom_eval,
165
+ simple_eval,
166
+ EvalBuilder,
167
+ )
168
+
169
+ # Unified API
170
+ from .evaluator import (
171
+ FrameworkEvaluator,
172
+ EvaluatorResult,
173
+ blocking_evaluator,
174
+ async_evaluator,
175
+ distributed_evaluator,
176
+ resilient_evaluator,
177
+ )
178
+
179
+ # Backwards-compatible alias used by tests and examples
180
+ Evaluator = FrameworkEvaluator
181
+
182
+ __all__ = [
183
+ # Version
184
+ "__version__",
185
+
186
+ # Types
187
+ "ExecutionMode",
188
+ "EvalStatus",
189
+ "FrameworkEvalResult",
190
+ "BatchEvalResult",
191
+ "EvalInputs",
192
+ "SpanAttributes",
193
+
194
+ # Context
195
+ "EvalContext",
196
+ "get_current_context",
197
+ "create_standalone_context",
198
+
199
+ # Protocols
200
+ "BaseEvaluation",
201
+ "EvalRegistry",
202
+ "register_evaluation",
203
+ "create_evaluation",
204
+
205
+ # Enrichment
206
+ "enrich_current_span",
207
+ "enrich_span",
208
+ "add_eval_event",
209
+ "get_current_span",
210
+ "is_span_recording",
211
+ "flatten_attributes",
212
+ "SpanEnricher",
213
+
214
+ # Registry
215
+ "SpanRegistry",
216
+ "register_span",
217
+ "get_span",
218
+ "unregister_span",
219
+ "get_registry",
220
+ "register_current_span",
221
+
222
+ # Propagation
223
+ "SpanContextPropagator",
224
+ "enrich_span_by_context",
225
+ "enrich_span_by_ids",
226
+ "add_event_by_context",
227
+ "ContextCarrier",
228
+ "propagate_context",
229
+ "propagate_context_lazy",
230
+
231
+ # Evaluators - Blocking
232
+ "BlockingEvaluator",
233
+ "blocking_evaluate",
234
+ # Evaluators - Non-Blocking
235
+ "NonBlockingEvaluator",
236
+ "non_blocking_evaluate",
237
+ "EvalFuture",
238
+ "BatchEvalFuture",
239
+ "EvalResultAggregator",
240
+ # Backends
241
+ "Backend",
242
+ "BackendConfig",
243
+ "TaskHandle",
244
+ "TaskStatus",
245
+ "ThreadPoolBackend",
246
+ "ThreadPoolConfig",
247
+ # Resilience
248
+ "ResilientBackend",
249
+ "ResilienceConfig",
250
+ "CircuitBreakerConfig",
251
+ "RateLimitConfig",
252
+ "RetryConfig",
253
+ "DegradationConfig",
254
+ "HealthCheckConfig",
255
+ "wrap_backend",
256
+ # Built-in evals + builder
257
+ "custom_eval",
258
+ "simple_eval",
259
+ "EvalBuilder",
260
+ # Unified API
261
+ "FrameworkEvaluator",
262
+ "EvaluatorResult",
263
+ "blocking_evaluator",
264
+ "async_evaluator",
265
+ "distributed_evaluator",
266
+ "resilient_evaluator",
267
+ ]
@@ -0,0 +1,33 @@
1
+ # Eval runner image for the Kubernetes backend.
2
+ #
3
+ # This image is used by KubernetesBackend to execute serialized evaluation
4
+ # tasks inside Kubernetes Jobs. It contains cloudpickle for deserialization
5
+ # and a minimal Python runtime.
6
+ #
7
+ # Build:
8
+ # docker build -f Dockerfile.eval-runner -t fi-eval-runner:latest .
9
+ #
10
+ # Usage:
11
+ # from fi.evals.framework.backends import KubernetesBackend, KubernetesConfig
12
+ #
13
+ # backend = KubernetesBackend(KubernetesConfig(
14
+ # image="fi-eval-runner:latest",
15
+ # ))
16
+
17
+ FROM python:3.11-slim
18
+
19
+ LABEL maintainer="Future AGI" \
20
+ description="Eval runner for fi-evals Kubernetes backend"
21
+
22
+ # Install only the runtime dependency needed for task deserialization.
23
+ # cloudpickle is the industry-standard serializer used by Kubeflow, Ray, etc.
24
+ # It is only used here in trusted evaluation environments — never for untrusted input.
25
+ RUN pip install --no-cache-dir cloudpickle>=3.0
26
+
27
+ # Drop to non-root user for safety
28
+ RUN useradd --create-home evalrunner
29
+ USER evalrunner
30
+ WORKDIR /home/evalrunner
31
+
32
+ # The actual task code is injected via the EVAL_PAYLOAD env var at runtime.
33
+ # The KubernetesBackend sets the container command to execute the runner script.
@@ -0,0 +1,99 @@
1
+ """
2
+ Backend implementations for distributed evaluation.
3
+
4
+ Provides pluggable backends for running evaluations:
5
+ - ThreadPoolBackend: Local execution using thread pool (default)
6
+ - TemporalBackend: Durable workflows via Temporal (optional)
7
+ - CeleryBackend: Distributed tasks via Celery (optional)
8
+ - RayBackend: Distributed computing via Ray (optional)
9
+ - KubernetesBackend: Cloud-native jobs via Kubernetes (optional)
10
+
11
+ Container utilities for custom backends:
12
+ - serialize_task / parse_result_from_logs: serialization protocol
13
+ - RUNNER_SCRIPT / RUNNER_COMMAND / DEFAULT_IMAGE: container constants
14
+ - Dockerfile.eval-runner: pre-built eval runner image
15
+
16
+ Optional backends are lazily imported to avoid requiring their dependencies.
17
+ Install optional dependencies with:
18
+ pip install fi-evals[temporal] # For Temporal
19
+ pip install fi-evals[celery] # For Celery
20
+ pip install fi-evals[ray] # For Ray
21
+ pip install fi-evals[kubernetes] # For Kubernetes
22
+ """
23
+
24
+ from typing import TYPE_CHECKING
25
+
26
+ from .base import (
27
+ Backend,
28
+ BackendConfig,
29
+ TaskHandle,
30
+ TaskStatus,
31
+ )
32
+ from .thread_pool import ThreadPoolBackend, ThreadPoolConfig
33
+ from ._container import (
34
+ DEFAULT_IMAGE,
35
+ EVAL_PAYLOAD_ENV,
36
+ RUNNER_COMMAND,
37
+ RUNNER_SCRIPT,
38
+ parse_result_from_logs,
39
+ serialize_task,
40
+ )
41
+
42
+ # Type hints for lazy imports (only used by type checkers)
43
+ if TYPE_CHECKING:
44
+ from .temporal import TemporalBackend, TemporalConfig
45
+ from .celery_backend import CeleryBackend, CeleryConfig
46
+ from .ray_backend import RayBackend, RayConfig
47
+ from .kubernetes_backend import KubernetesBackend, KubernetesConfig
48
+
49
+ __all__ = [
50
+ # Base
51
+ "Backend",
52
+ "BackendConfig",
53
+ "TaskHandle",
54
+ "TaskStatus",
55
+ # Thread Pool
56
+ "ThreadPoolBackend",
57
+ "ThreadPoolConfig",
58
+ # Temporal (optional)
59
+ "TemporalBackend",
60
+ "TemporalConfig",
61
+ # Celery (optional)
62
+ "CeleryBackend",
63
+ "CeleryConfig",
64
+ # Ray (optional)
65
+ "RayBackend",
66
+ "RayConfig",
67
+ # Kubernetes (optional)
68
+ "KubernetesBackend",
69
+ "KubernetesConfig",
70
+ # Container utilities (for custom backends)
71
+ "DEFAULT_IMAGE",
72
+ "EVAL_PAYLOAD_ENV",
73
+ "RUNNER_COMMAND",
74
+ "RUNNER_SCRIPT",
75
+ "serialize_task",
76
+ "parse_result_from_logs",
77
+ ]
78
+
79
+ # Lazy imports for optional backends
80
+ _LAZY_IMPORTS = {
81
+ "TemporalBackend": (".temporal", "TemporalBackend"),
82
+ "TemporalConfig": (".temporal", "TemporalConfig"),
83
+ "CeleryBackend": (".celery_backend", "CeleryBackend"),
84
+ "CeleryConfig": (".celery_backend", "CeleryConfig"),
85
+ "RayBackend": (".ray_backend", "RayBackend"),
86
+ "RayConfig": (".ray_backend", "RayConfig"),
87
+ "KubernetesBackend": (".kubernetes_backend", "KubernetesBackend"),
88
+ "KubernetesConfig": (".kubernetes_backend", "KubernetesConfig"),
89
+ }
90
+
91
+
92
+ def __getattr__(name: str):
93
+ """Lazy import for optional backend dependencies."""
94
+ if name in _LAZY_IMPORTS:
95
+ module_name, attr_name = _LAZY_IMPORTS[name]
96
+ import importlib
97
+ module = importlib.import_module(module_name, __package__)
98
+ return getattr(module, attr_name)
99
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,141 @@
1
+ """
2
+ Shared container utilities for container-based backends.
3
+
4
+ Provides the serialization protocol, runner script, and result parsing
5
+ used by any backend that executes eval tasks inside containers (Kubernetes,
6
+ Docker, ECS, Nomad, etc.).
7
+
8
+ Custom backends can reuse these utilities::
9
+
10
+ from fi.evals.framework.backends._container import (
11
+ serialize_task,
12
+ parse_result_from_logs,
13
+ RUNNER_SCRIPT,
14
+ EVAL_PAYLOAD_ENV,
15
+ DEFAULT_IMAGE,
16
+ )
17
+
18
+ # Serialize (fn, args, kwargs) into a base64 string
19
+ payload = serialize_task(my_func, (arg1,), {"key": "val"})
20
+
21
+ # Pass payload as env var EVAL_PAYLOAD into any container that runs RUNNER_SCRIPT
22
+
23
+ # After container finishes, parse the JSON result from its stdout
24
+ result = parse_result_from_logs(container_stdout)
25
+
26
+ Security note:
27
+ cloudpickle is the industry-standard serializer used by Kubeflow Pipelines,
28
+ Ray, Dask, etc. It is only used here in trusted evaluation environments
29
+ (your own code running on your own infrastructure) — never for untrusted input.
30
+ """
31
+
32
+ import base64
33
+ import json
34
+ import logging
35
+ from typing import Any, Callable, Dict, Optional
36
+
37
+ logger = logging.getLogger(__name__)
38
+
39
+ # ---------------------------------------------------------------------------
40
+ # Constants
41
+ # ---------------------------------------------------------------------------
42
+
43
+ #: Default eval runner image. Build from ``Dockerfile.eval-runner``.
44
+ DEFAULT_IMAGE: str = "fi-eval-runner:latest"
45
+
46
+ #: Environment variable name for the serialized task payload.
47
+ EVAL_PAYLOAD_ENV: str = "EVAL_PAYLOAD"
48
+
49
+ #: Python bootstrap script to inject into containers.
50
+ #: Reads EVAL_PAYLOAD env var, deserializes with cloudpickle, runs the
51
+ #: function, and prints a JSON result line to stdout.
52
+ RUNNER_SCRIPT: str = """
53
+ import base64, cloudpickle, json, sys, traceback
54
+ try:
55
+ import os
56
+ payload = base64.b64decode(os.environ["EVAL_PAYLOAD"])
57
+ fn, args, kwargs = cloudpickle.loads(payload)
58
+ result = fn(*args, **kwargs)
59
+ print(json.dumps({"status": "success", "result": result}))
60
+ except Exception:
61
+ tb = traceback.format_exc()
62
+ print(json.dumps({"status": "error", "error": tb}))
63
+ sys.exit(1)
64
+ """
65
+
66
+ #: Container command that executes the runner script.
67
+ RUNNER_COMMAND: list = ["python", "-c", RUNNER_SCRIPT]
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Serialization
72
+ # ---------------------------------------------------------------------------
73
+
74
+ def serialize_task(
75
+ fn: Callable,
76
+ args: tuple = (),
77
+ kwargs: Optional[Dict[str, Any]] = None,
78
+ ) -> str:
79
+ """
80
+ Serialize a task (function + arguments) into a base64-encoded string.
81
+
82
+ The result is safe to pass as an environment variable to a container
83
+ running ``RUNNER_SCRIPT``.
84
+
85
+ Args:
86
+ fn: The function to execute.
87
+ args: Positional arguments.
88
+ kwargs: Keyword arguments.
89
+
90
+ Returns:
91
+ Base64-encoded string of the cloudpickle-serialized payload.
92
+
93
+ Raises:
94
+ ImportError: If cloudpickle is not installed.
95
+ """
96
+ import cloudpickle
97
+
98
+ kwargs = kwargs or {}
99
+ payload = cloudpickle.dumps((fn, args, kwargs))
100
+ return base64.b64encode(payload).decode("utf-8")
101
+
102
+
103
+ # ---------------------------------------------------------------------------
104
+ # Result parsing
105
+ # ---------------------------------------------------------------------------
106
+
107
+ def parse_result_from_logs(logs: str) -> Any:
108
+ """
109
+ Parse the task result from container stdout logs.
110
+
111
+ The runner script prints a JSON line as its last output. This function
112
+ walks the log lines in reverse to find and parse that JSON.
113
+
114
+ Args:
115
+ logs: The full stdout text from the container.
116
+
117
+ Returns:
118
+ The deserialized result value.
119
+
120
+ Raises:
121
+ RuntimeError: If the logs contain an error result, are empty,
122
+ or don't contain a valid JSON result line.
123
+ """
124
+ lines = logs.strip().splitlines()
125
+ if not lines:
126
+ raise RuntimeError("Empty container logs — no result found")
127
+
128
+ for line in reversed(lines):
129
+ line = line.strip()
130
+ if not line:
131
+ continue
132
+ try:
133
+ data = json.loads(line)
134
+ if data.get("status") == "success":
135
+ return data["result"]
136
+ elif data.get("status") == "error":
137
+ raise RuntimeError(data["error"])
138
+ except json.JSONDecodeError:
139
+ continue
140
+
141
+ raise RuntimeError("No JSON result found in container logs")
@@ -0,0 +1,145 @@
1
+ """
2
+ Utilities for optional dependency handling.
3
+
4
+ Provides helpers for gracefully handling missing optional dependencies
5
+ with clear error messages guiding users to install them.
6
+ """
7
+
8
+ from typing import Any, Optional
9
+
10
+
11
+ class OptionalDependency:
12
+ """
13
+ Wrapper for optional imports with helpful error messages.
14
+
15
+ Example:
16
+ TEMPORAL = OptionalDependency("temporalio", "pip install temporalio")
17
+
18
+ # Check availability
19
+ if TEMPORAL.is_available():
20
+ client = TEMPORAL.require()
21
+
22
+ # Or just require (raises if not available)
23
+ temporalio = TEMPORAL.require()
24
+ """
25
+
26
+ def __init__(
27
+ self,
28
+ module_name: str,
29
+ install_hint: str,
30
+ extra_name: Optional[str] = None,
31
+ ):
32
+ """
33
+ Initialize optional dependency wrapper.
34
+
35
+ Args:
36
+ module_name: The Python module name to import
37
+ install_hint: Installation command to show in error message
38
+ extra_name: Optional extra name for pip install (e.g., 'temporal')
39
+ """
40
+ self.module_name = module_name
41
+ self.install_hint = install_hint
42
+ self.extra_name = extra_name or module_name
43
+ self._module: Optional[Any] = None
44
+ self._checked = False
45
+ self._available = False
46
+
47
+ def is_available(self) -> bool:
48
+ """Check if the dependency is installed."""
49
+ if not self._checked:
50
+ try:
51
+ self._module = __import__(self.module_name)
52
+ self._available = True
53
+ except ImportError:
54
+ self._available = False
55
+ self._checked = True
56
+ return self._available
57
+
58
+ def require(self) -> Any:
59
+ """
60
+ Require the dependency, raising ImportError if not available.
61
+
62
+ Returns:
63
+ The imported module
64
+
65
+ Raises:
66
+ ImportError: If dependency is not installed
67
+ """
68
+ if not self.is_available():
69
+ raise ImportError(
70
+ f"'{self.module_name}' is required but not installed. "
71
+ f"Install it with: {self.install_hint}\n"
72
+ f"Or install the extra: pip install fi-evals[{self.extra_name}]"
73
+ )
74
+ return self._module
75
+
76
+ def import_from(self, *names: str) -> tuple:
77
+ """
78
+ Import specific names from the module.
79
+
80
+ Args:
81
+ *names: Names to import from the module
82
+
83
+ Returns:
84
+ Tuple of imported objects
85
+
86
+ Raises:
87
+ ImportError: If dependency is not installed
88
+ """
89
+ module = self.require()
90
+ result = []
91
+ for name in names:
92
+ parts = name.split(".")
93
+ obj = module
94
+ for part in parts:
95
+ obj = getattr(obj, part)
96
+ result.append(obj)
97
+ return tuple(result) if len(result) > 1 else result[0]
98
+
99
+
100
+ # Pre-configured optional dependencies
101
+ TEMPORAL = OptionalDependency(
102
+ "temporalio",
103
+ "pip install temporalio",
104
+ "temporal",
105
+ )
106
+
107
+ CELERY = OptionalDependency(
108
+ "celery",
109
+ "pip install 'celery[redis]'",
110
+ "celery",
111
+ )
112
+
113
+ RAY = OptionalDependency(
114
+ "ray",
115
+ "pip install 'ray[default]'",
116
+ "ray",
117
+ )
118
+
119
+ KUBERNETES = OptionalDependency(
120
+ "kubernetes",
121
+ "pip install kubernetes",
122
+ "kubernetes",
123
+ )
124
+
125
+
126
+ def check_dependency(name: str) -> bool:
127
+ """
128
+ Check if a named dependency is available.
129
+
130
+ Args:
131
+ name: Dependency name ('temporal', 'celery', 'ray', 'kubernetes')
132
+
133
+ Returns:
134
+ True if available, False otherwise
135
+ """
136
+ deps = {
137
+ "temporal": TEMPORAL,
138
+ "celery": CELERY,
139
+ "ray": RAY,
140
+ "kubernetes": KUBERNETES,
141
+ }
142
+ dep = deps.get(name.lower())
143
+ if dep:
144
+ return dep.is_available()
145
+ return False