agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,128 @@
1
+ """Feedback retrieval and few-shot formatting.
2
+
3
+ Retrieves semantically similar feedback from the store and formats it
4
+ as few-shot examples for the LLM judge pipeline.
5
+ """
6
+
7
+ import json
8
+ import logging
9
+ from typing import Any, Dict, List, Optional
10
+
11
+ from .store import FeedbackStore
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ class FeedbackRetriever:
17
+ """Retrieves semantically similar feedback and formats as few-shot examples.
18
+
19
+ This is the bridge between the feedback store and the LLM judge pipeline.
20
+ When a metric is run with a feedback store, the retriever:
21
+
22
+ 1. Builds an embedding query from the current inputs
23
+ 2. Searches the store for similar past feedback entries
24
+ 3. Converts matching entries to the few_shot_examples format
25
+ expected by CustomLLMJudge's Jinja2 template
26
+
27
+ Args:
28
+ store: The FeedbackStore to search.
29
+ max_examples: Maximum number of few-shot examples to inject. Default 3.
30
+ """
31
+
32
+ def __init__(
33
+ self,
34
+ store: FeedbackStore,
35
+ max_examples: int = 3,
36
+ ):
37
+ self.store = store
38
+ self.max_examples = max_examples
39
+
40
+ def build_query_text(self, metric_name: str, inputs: Dict[str, Any]) -> str:
41
+ """Build a query string from inputs for semantic search.
42
+
43
+ Uses the same concatenation strategy as FeedbackEntry.to_embedding_text()
44
+ to ensure query-document alignment.
45
+ """
46
+ parts = [f"metric: {metric_name}"]
47
+ for key in ("output", "response", "context", "input", "query"):
48
+ val = inputs.get(key)
49
+ if val:
50
+ text = val if isinstance(val, str) else json.dumps(val, default=str)
51
+ parts.append(f"{key}: {text[:500]}")
52
+ return "\n".join(parts)
53
+
54
+ def retrieve_few_shot_examples(
55
+ self,
56
+ metric_name: str,
57
+ inputs: Dict[str, Any],
58
+ ) -> List[Dict[str, Any]]:
59
+ """Retrieve few-shot examples from feedback store.
60
+
61
+ Returns a list of dicts in the format expected by
62
+ CustomLLMJudge's config["few_shot_examples"]:
63
+
64
+ [{"inputs": {...}, "output": '{"score": 0.8, "reason": "..."}'}]
65
+
66
+ Args:
67
+ metric_name: The metric being run.
68
+ inputs: The current inputs.
69
+
70
+ Returns:
71
+ List of few-shot example dicts, possibly empty if no feedback exists.
72
+ """
73
+ if self.store.count(metric_name) == 0:
74
+ return []
75
+
76
+ query_text = self.build_query_text(metric_name, inputs)
77
+
78
+ similar_entries = self.store.query_similar(
79
+ metric_name=metric_name,
80
+ text=query_text,
81
+ n_results=self.max_examples,
82
+ )
83
+
84
+ if not similar_entries:
85
+ return []
86
+
87
+ examples = []
88
+ for entry in similar_entries:
89
+ # Only include entries where the developer provided a correction
90
+ if entry.correct_score is None and not entry.correct_reason:
91
+ continue
92
+ examples.append(entry.to_few_shot())
93
+
94
+ if examples:
95
+ logger.debug(
96
+ f"Retrieved {len(examples)} feedback examples for '{metric_name}'"
97
+ )
98
+
99
+ return examples
100
+
101
+ def inject_into_config(
102
+ self,
103
+ metric_name: str,
104
+ inputs: Dict[str, Any],
105
+ config: Optional[Dict[str, Any]] = None,
106
+ ) -> Dict[str, Any]:
107
+ """Retrieve feedback and merge into a config dict for LLMEngine/CustomLLMJudge.
108
+
109
+ This is the primary integration point. Call this before passing config
110
+ to LLMEngine.run() to inject few-shot examples.
111
+
112
+ Args:
113
+ metric_name: Metric name.
114
+ inputs: Current inputs.
115
+ config: Existing config dict (will not be mutated).
116
+
117
+ Returns:
118
+ New config dict with few_shot_examples populated.
119
+ """
120
+ config = dict(config or {})
121
+
122
+ examples = self.retrieve_few_shot_examples(metric_name, inputs)
123
+ if examples:
124
+ # Merge with any existing few-shot examples
125
+ existing = config.get("few_shot_examples", [])
126
+ config["few_shot_examples"] = existing + examples
127
+
128
+ return config
@@ -0,0 +1,272 @@
1
+ """Feedback storage backends.
2
+
3
+ Provides abstract FeedbackStore and two implementations:
4
+ - InMemoryFeedbackStore: for testing and small-scale usage
5
+ - ChromaFeedbackStore: for production with semantic vector search
6
+ """
7
+
8
+ import json
9
+ import logging
10
+ from abc import ABC, abstractmethod
11
+ from typing import Any, Dict, List, Optional
12
+
13
+ from .types import FeedbackEntry
14
+
15
+ logger = logging.getLogger(__name__)
16
+
17
+
18
+ class FeedbackStore(ABC):
19
+ """Abstract base class for feedback persistence."""
20
+
21
+ @abstractmethod
22
+ def add(self, entry: FeedbackEntry) -> str:
23
+ """Store a feedback entry. Returns the entry ID."""
24
+ ...
25
+
26
+ @abstractmethod
27
+ def query_similar(
28
+ self,
29
+ metric_name: str,
30
+ text: str,
31
+ n_results: int = 5,
32
+ ) -> List[FeedbackEntry]:
33
+ """Find feedback entries semantically similar to the given text,
34
+ filtered by metric_name."""
35
+ ...
36
+
37
+ @abstractmethod
38
+ def get_by_metric(self, metric_name: str, limit: int = 100) -> List[FeedbackEntry]:
39
+ """Get all feedback entries for a specific metric."""
40
+ ...
41
+
42
+ @abstractmethod
43
+ def count(self, metric_name: Optional[str] = None) -> int:
44
+ """Count entries, optionally filtered by metric_name."""
45
+ ...
46
+
47
+ @abstractmethod
48
+ def delete(self, entry_id: str) -> bool:
49
+ """Delete a feedback entry by ID."""
50
+ ...
51
+
52
+
53
+ class InMemoryFeedbackStore(FeedbackStore):
54
+ """In-memory feedback store for testing and small-scale usage.
55
+
56
+ No vector search -- falls back to recency-based retrieval.
57
+ Suitable for unit tests and quick experimentation.
58
+ """
59
+
60
+ def __init__(self):
61
+ self._entries: Dict[str, FeedbackEntry] = {}
62
+
63
+ def add(self, entry: FeedbackEntry) -> str:
64
+ self._entries[entry.id] = entry
65
+ return entry.id
66
+
67
+ def query_similar(
68
+ self,
69
+ metric_name: str,
70
+ text: str,
71
+ n_results: int = 5,
72
+ ) -> List[FeedbackEntry]:
73
+ # No semantic search -- return most recent entries for this metric
74
+ filtered = [
75
+ e for e in self._entries.values()
76
+ if e.eval_name == metric_name
77
+ ]
78
+ filtered.sort(key=lambda e: e.created_at, reverse=True)
79
+ return filtered[:n_results]
80
+
81
+ def get_by_metric(self, metric_name: str, limit: int = 100) -> List[FeedbackEntry]:
82
+ return [
83
+ e for e in self._entries.values()
84
+ if e.eval_name == metric_name
85
+ ][:limit]
86
+
87
+ def count(self, metric_name: Optional[str] = None) -> int:
88
+ if metric_name:
89
+ return sum(1 for e in self._entries.values() if e.eval_name == metric_name)
90
+ return len(self._entries)
91
+
92
+ def delete(self, entry_id: str) -> bool:
93
+ return self._entries.pop(entry_id, None) is not None
94
+
95
+
96
+ class ChromaFeedbackStore(FeedbackStore):
97
+ """ChromaDB-backed feedback store with semantic vector search.
98
+
99
+ Supports two modes:
100
+ - Local: in-process persistent ChromaDB (default)
101
+ - Service: connects to a remote ChromaDB server (e.g. Docker container)
102
+
103
+ Embedding is handled by ChromaDB's built-in default embedding function
104
+ (all-MiniLM-L6-v2 via sentence-transformers) OR via a LiteLLM embedding
105
+ function for API-based embeddings.
106
+
107
+ Args:
108
+ host: ChromaDB server host. None = local persistent mode.
109
+ port: ChromaDB server port. Default 8000.
110
+ persist_directory: Local storage path (local mode only).
111
+ Default "~/.fi/feedback/chroma".
112
+ collection_prefix: Prefix for ChromaDB collection names.
113
+ embedding_model: LiteLLM model string for embeddings.
114
+ None = use ChromaDB's default (sentence-transformers).
115
+ """
116
+
117
+ def __init__(
118
+ self,
119
+ host: Optional[str] = None,
120
+ port: int = 8000,
121
+ persist_directory: Optional[str] = None,
122
+ collection_prefix: str = "fi_feedback",
123
+ embedding_model: Optional[str] = None,
124
+ ):
125
+ try:
126
+ import chromadb
127
+ except ImportError:
128
+ raise ImportError(
129
+ "chromadb is required for ChromaFeedbackStore. "
130
+ "Install it with: pip install ai-evaluation[feedback]"
131
+ )
132
+
133
+ self._collection_prefix = collection_prefix
134
+ self._embedding_model = embedding_model
135
+
136
+ # Initialize ChromaDB client
137
+ if host:
138
+ self._client = chromadb.HttpClient(host=host, port=port)
139
+ logger.info(f"Connected to ChromaDB server at {host}:{port}")
140
+ else:
141
+ import os
142
+ path = persist_directory or os.path.expanduser("~/.fi/feedback/chroma")
143
+ os.makedirs(path, exist_ok=True)
144
+ self._client = chromadb.PersistentClient(path=path)
145
+ logger.info(f"Using local ChromaDB at {path}")
146
+
147
+ # Set up embedding function
148
+ self._embedding_fn = None
149
+ if embedding_model:
150
+ self._embedding_fn = self._make_litellm_embedding_fn(embedding_model)
151
+
152
+ @staticmethod
153
+ def _make_litellm_embedding_fn(model: str):
154
+ """Create a ChromaDB-compatible embedding function using LiteLLM."""
155
+ from chromadb.api.types import EmbeddingFunction, Documents, Embeddings
156
+ import litellm
157
+
158
+ class LiteLLMEmbedding(EmbeddingFunction):
159
+ def __call__(self, input: Documents) -> Embeddings:
160
+ response = litellm.embedding(model=model, input=input)
161
+ return [item["embedding"] for item in response.data]
162
+
163
+ return LiteLLMEmbedding()
164
+
165
+ def _get_collection(self, metric_name: str):
166
+ """Get or create a ChromaDB collection for a specific metric."""
167
+ name = f"{self._collection_prefix}_{metric_name}".replace(".", "_")
168
+ kwargs: Dict[str, Any] = {"name": name}
169
+ if self._embedding_fn:
170
+ kwargs["embedding_function"] = self._embedding_fn
171
+ return self._client.get_or_create_collection(**kwargs)
172
+
173
+ def add(self, entry: FeedbackEntry) -> str:
174
+ collection = self._get_collection(entry.eval_name)
175
+ collection.add(
176
+ ids=[entry.id],
177
+ documents=[entry.to_embedding_text()],
178
+ metadatas=[{
179
+ "metric_name": entry.eval_name,
180
+ "original_score": entry.original_score or 0.0,
181
+ "correct_score": entry.correct_score if entry.correct_score is not None else -1.0,
182
+ "correct_reason": entry.correct_reason[:1000],
183
+ "inputs_json": json.dumps(entry.inputs, default=str)[:4000],
184
+ "original_reason": entry.original_reason[:1000],
185
+ "created_at": entry.created_at.isoformat(),
186
+ }],
187
+ )
188
+ return entry.id
189
+
190
+ def query_similar(
191
+ self,
192
+ metric_name: str,
193
+ text: str,
194
+ n_results: int = 5,
195
+ ) -> List[FeedbackEntry]:
196
+ collection = self._get_collection(metric_name)
197
+
198
+ if collection.count() == 0:
199
+ return []
200
+
201
+ actual_n = min(n_results, collection.count())
202
+
203
+ results = collection.query(
204
+ query_texts=[text],
205
+ n_results=actual_n,
206
+ )
207
+
208
+ entries = []
209
+ for i, meta in enumerate(results["metadatas"][0]):
210
+ inputs = {}
211
+ try:
212
+ inputs = json.loads(meta.get("inputs_json", "{}"))
213
+ except (json.JSONDecodeError, TypeError):
214
+ pass
215
+
216
+ correct_score_val = meta.get("correct_score", -1.0)
217
+ entry = FeedbackEntry(
218
+ id=results["ids"][0][i],
219
+ eval_name=meta.get("metric_name", metric_name),
220
+ inputs=inputs,
221
+ original_score=meta.get("original_score"),
222
+ original_reason=meta.get("original_reason", ""),
223
+ correct_score=correct_score_val if correct_score_val >= 0 else None,
224
+ correct_reason=meta.get("correct_reason", ""),
225
+ )
226
+ entries.append(entry)
227
+
228
+ return entries
229
+
230
+ def get_by_metric(self, metric_name: str, limit: int = 100) -> List[FeedbackEntry]:
231
+ collection = self._get_collection(metric_name)
232
+ if collection.count() == 0:
233
+ return []
234
+ results = collection.get(limit=limit)
235
+ entries = []
236
+ for i, meta in enumerate(results["metadatas"]):
237
+ inputs = {}
238
+ try:
239
+ inputs = json.loads(meta.get("inputs_json", "{}"))
240
+ except (json.JSONDecodeError, TypeError):
241
+ pass
242
+ correct_score_val = meta.get("correct_score", -1.0)
243
+ entry = FeedbackEntry(
244
+ id=results["ids"][i],
245
+ eval_name=meta.get("metric_name", metric_name),
246
+ inputs=inputs,
247
+ original_score=meta.get("original_score"),
248
+ original_reason=meta.get("original_reason", ""),
249
+ correct_score=correct_score_val if correct_score_val >= 0 else None,
250
+ correct_reason=meta.get("correct_reason", ""),
251
+ )
252
+ entries.append(entry)
253
+ return entries
254
+
255
+ def count(self, metric_name: Optional[str] = None) -> int:
256
+ if metric_name:
257
+ return self._get_collection(metric_name).count()
258
+ total = 0
259
+ for col in self._client.list_collections():
260
+ if col.name.startswith(self._collection_prefix):
261
+ total += col.count()
262
+ return total
263
+
264
+ def delete(self, entry_id: str) -> bool:
265
+ for col in self._client.list_collections():
266
+ if col.name.startswith(self._collection_prefix):
267
+ try:
268
+ col.delete(ids=[entry_id])
269
+ return True
270
+ except Exception:
271
+ continue
272
+ return False
@@ -0,0 +1,99 @@
1
+ """Type definitions for the Feedback Loop system.
2
+
3
+ Provides dataclasses for storing developer feedback on evaluation results,
4
+ calibration profiles, and aggregate statistics.
5
+ """
6
+
7
+ import json
8
+ import uuid
9
+ from dataclasses import dataclass, field
10
+ from datetime import datetime, timezone
11
+ from typing import Any, Dict, List, Optional
12
+
13
+
14
+ @dataclass
15
+ class FeedbackEntry:
16
+ """A single piece of developer feedback on an evaluation result."""
17
+
18
+ id: str = field(default_factory=lambda: str(uuid.uuid4()))
19
+
20
+ # What was evaluated
21
+ eval_name: str = ""
22
+ inputs: Dict[str, Any] = field(default_factory=dict)
23
+
24
+ # What the system produced
25
+ original_score: Optional[float] = None
26
+ original_reason: str = ""
27
+ original_passed: Optional[bool] = None
28
+
29
+ # What the developer says is correct
30
+ correct_score: Optional[float] = None
31
+ correct_passed: Optional[bool] = None
32
+ correct_reason: str = ""
33
+
34
+ # Metadata
35
+ tags: List[str] = field(default_factory=list)
36
+ created_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
37
+ metadata: Dict[str, Any] = field(default_factory=dict)
38
+
39
+ def to_few_shot(self) -> Dict[str, Any]:
40
+ """Convert to the format expected by CustomLLMJudge few_shot_examples.
41
+
42
+ Returns dict matching the template schema:
43
+ {"inputs": {...}, "output": "<json with score/reason>"}
44
+ """
45
+ output_dict = {
46
+ "score": self.correct_score if self.correct_score is not None else self.original_score,
47
+ "reason": self.correct_reason or self.original_reason,
48
+ }
49
+ return {
50
+ "inputs": self.inputs,
51
+ "output": json.dumps(output_dict),
52
+ }
53
+
54
+ def to_embedding_text(self) -> str:
55
+ """Create a text representation for embedding.
56
+
57
+ Concatenates key input fields into a string suitable for
58
+ semantic embedding. Prioritizes output, context, and input fields.
59
+ """
60
+ parts = [f"metric: {self.eval_name}"]
61
+ for key in ("output", "response", "context", "input", "query"):
62
+ val = self.inputs.get(key)
63
+ if val:
64
+ text = val if isinstance(val, str) else json.dumps(val, default=str)
65
+ parts.append(f"{key}: {text[:500]}")
66
+ return "\n".join(parts)
67
+
68
+
69
+ @dataclass
70
+ class CalibrationProfile:
71
+ """Optimized threshold settings for a metric based on feedback."""
72
+
73
+ eval_name: str
74
+ optimal_threshold: float
75
+ sample_size: int
76
+ accuracy_at_threshold: float # % of feedback entries that agree at this threshold
77
+
78
+ # Distribution stats
79
+ score_mean: float = 0.0
80
+ score_std: float = 0.0
81
+
82
+ # Confusion matrix at the optimal threshold
83
+ true_positives: int = 0
84
+ false_positives: int = 0
85
+ true_negatives: int = 0
86
+ false_negatives: int = 0
87
+
88
+ metadata: Dict[str, Any] = field(default_factory=dict)
89
+
90
+
91
+ @dataclass
92
+ class FeedbackStats:
93
+ """Aggregate statistics for feedback on a given metric."""
94
+
95
+ eval_name: str
96
+ total_entries: int = 0
97
+ agreement_rate: float = 0.0 # How often original == correct
98
+ avg_score_delta: float = 0.0 # avg(correct_score - original_score)
99
+ score_distribution: Dict[str, int] = field(default_factory=dict) # buckets
@@ -0,0 +1,79 @@
1
+ # Evaluation Framework
2
+
3
+ A scalable evaluation infrastructure for AI systems with support for blocking, non-blocking, and distributed execution modes.
4
+
5
+ ## Quick Start
6
+
7
+ ```python
8
+ from fi.evals import FrameworkEvaluator, ExecutionMode
9
+ from fi.evals.framework.evals import CoherenceEval, ActionSafetyEval
10
+
11
+ # Create an evaluator with multiple evaluations
12
+ evaluator = FrameworkEvaluator(
13
+ evaluations=[
14
+ CoherenceEval(),
15
+ ActionSafetyEval(),
16
+ ],
17
+ mode=ExecutionMode.BLOCKING,
18
+ )
19
+
20
+ # Run evaluations
21
+ result = evaluator.run({
22
+ "response": "Paris is the capital of France. It is in Western Europe.",
23
+ "trajectory": [
24
+ {"type": "tool_call", "tool": "search", "args": "Paris facts"},
25
+ ],
26
+ })
27
+
28
+ # Check results
29
+ for r in result.results:
30
+ print(f"{r.eval_name}: score={r.value.score:.2f}, passed={r.value.passed}")
31
+ ```
32
+
33
+ ## Execution Modes
34
+
35
+ | Mode | Use Case | Latency Impact |
36
+ |------|----------|----------------|
37
+ | `BLOCKING` | Development, testing, sync workflows | Full evaluation time |
38
+ | `NON_BLOCKING` | Production, real-time applications | Zero (async) |
39
+ | `DISTRIBUTED` | Batch processing, high throughput | Zero (remote) |
40
+
41
+ ## Available Evaluations
42
+
43
+ ### Semantic Evaluations
44
+ - `CoherenceEval` - Check text coherence
45
+
46
+ ### Agentic Evaluations
47
+ - `ActionSafetyEval` - Safety scanning
48
+ - `ReasoningQualityEval` - Reasoning quality
49
+
50
+ ### Custom Evaluation Builders
51
+ - `EvalBuilder` - Fluent builder pattern
52
+ - `@custom_eval` - Decorator for functions
53
+ - `simple_eval()` - Score-based evaluation
54
+ - `comparison_eval()` - Compare two fields
55
+ - `threshold_eval()` - Min/max thresholds
56
+ - `pattern_match_eval()` - Regex patterns
57
+
58
+ ## OpenTelemetry Integration
59
+
60
+ All evaluations automatically generate span attributes compatible with OpenTelemetry:
61
+
62
+ ```python
63
+ from fi.evals.framework import register_current_span, async_evaluator
64
+
65
+ # Register span for cross-thread enrichment
66
+ with tracer.start_as_current_span("llm_call") as span:
67
+ register_current_span()
68
+
69
+ response = llm.complete(prompt)
70
+ evaluator.run({"response": response}) # Enriches span automatically
71
+
72
+ return response
73
+
74
+ # Span attributes include:
75
+ # eval.coherence.score
76
+ # eval.coherence.passed
77
+ # eval.action_safety.score
78
+ # etc.
79
+ ```