agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/evals/execution.py ADDED
@@ -0,0 +1,168 @@
1
+ """
2
+ Execution handles for async eval and composite runs.
3
+
4
+ An ``Execution`` is a lightweight view into a (possibly still-running) eval
5
+ on the backend (for single evals) or a background thread (for composite
6
+ evals, which the backend runs synchronously).
7
+
8
+ Typical usage::
9
+
10
+ from fi.evals import Evaluator, EvalTemplateManager
11
+
12
+ ev = Evaluator()
13
+ mgr = EvalTemplateManager()
14
+
15
+ # --- Single eval: real backend async via is_async=True ---
16
+ handle = ev.submit("tone", {"output": "I love this!"})
17
+ handle.wait() # polls until completion
18
+ print(handle.result.output) # -> "love"
19
+
20
+ # --- Composite eval: SDK-side threaded execution ---
21
+ handle = mgr.submit_composite(composite_id, mapping={"output": "Hi!"})
22
+ handle.wait()
23
+ print(handle.result["aggregate_score"])
24
+
25
+ # --- Resumable by ID (single eval only) ---
26
+ other_handle = ev.get_execution(handle.id)
27
+ other_handle.wait()
28
+
29
+ Note on composite executions: the handle lives in a background thread
30
+ inside the calling process. If the process dies or the handle is dropped,
31
+ the in-flight work is lost. Use single-eval async submissions when you
32
+ need cross-process resumability.
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ import time
38
+ from dataclasses import dataclass, field
39
+ from typing import Any, Callable, Dict, Optional
40
+
41
+
42
+ class ExecutionError(Exception):
43
+ """Raised when an Execution that finished in the ``failed`` state is awaited."""
44
+
45
+
46
+ # Backend eval_status → SDK-normalized status.
47
+ _STATUS_MAP = {
48
+ "pending": "pending",
49
+ "PENDING": "pending",
50
+ "processing": "processing",
51
+ "PROCESSING": "processing",
52
+ "running": "processing",
53
+ "completed": "completed",
54
+ "COMPLETED": "completed",
55
+ "failed": "failed",
56
+ "FAILED": "failed",
57
+ }
58
+
59
+
60
+ def _normalize_status(raw: Optional[str]) -> str:
61
+ if not raw:
62
+ return "pending"
63
+ return _STATUS_MAP.get(raw, raw.lower())
64
+
65
+
66
+ @dataclass
67
+ class Execution:
68
+ """
69
+ Handle to an in-flight or completed eval execution.
70
+
71
+ Attributes:
72
+ id: Execution identifier. For single evals this is the server-side
73
+ ``eval_id`` (a UUID) and is resumable from any process via
74
+ :py:meth:`fi.evals.Evaluator.get_execution`. For composite
75
+ evals this is a client-side UUID — see the module docstring
76
+ for the caveat.
77
+ kind: ``"eval"`` for a single eval execution, ``"composite"`` for
78
+ a composite one.
79
+ status: ``"pending"`` | ``"processing"`` | ``"completed"`` |
80
+ ``"failed"``.
81
+ result: Populated once ``status == "completed"``. An ``EvalResult``
82
+ for single evals; a dict matching the composite execute
83
+ response for composite ones.
84
+ error_message: Populated when the execution failed.
85
+ error_localizer: Populated for single evals when error
86
+ localization was enabled and the analysis is available.
87
+ """
88
+
89
+ id: str
90
+ kind: str
91
+ status: str = "pending"
92
+ result: Any = None
93
+ error_message: Optional[str] = None
94
+ error_localizer: Optional[Dict[str, Any]] = None
95
+
96
+ # Closure that (re)fetches the latest state. Set by the factory
97
+ # method on Evaluator / EvalTemplateManager. Excluded from repr.
98
+ _refresher: Optional[Callable[[], "Execution"]] = field(
99
+ default=None, repr=False, compare=False
100
+ )
101
+
102
+ def is_done(self) -> bool:
103
+ """Return True if status is terminal (``completed`` or ``failed``)."""
104
+ return self.status in ("completed", "failed")
105
+
106
+ def refresh(self) -> "Execution":
107
+ """
108
+ Re-fetch the latest state from the source (backend for single
109
+ evals, background thread for composites). Returns ``self``.
110
+ """
111
+ if self._refresher is None:
112
+ return self
113
+ updated = self._refresher()
114
+ self.status = updated.status
115
+ self.result = updated.result
116
+ self.error_message = updated.error_message
117
+ self.error_localizer = updated.error_localizer
118
+ return self
119
+
120
+ def wait(
121
+ self,
122
+ *,
123
+ timeout: float = 300.0,
124
+ poll_interval: float = 2.0,
125
+ raise_on_failure: bool = True,
126
+ ) -> "Execution":
127
+ """
128
+ Block until the execution reaches a terminal state, refreshing
129
+ every ``poll_interval`` seconds.
130
+
131
+ Args:
132
+ timeout: Maximum number of seconds to wait before giving up.
133
+ poll_interval: Seconds between refreshes.
134
+ raise_on_failure: If True (default) raise
135
+ :class:`ExecutionError` when the run finished in
136
+ ``"failed"`` state. If False, return the handle so the
137
+ caller can inspect ``error_message`` themselves.
138
+
139
+ Raises:
140
+ TimeoutError: If the execution did not reach a terminal
141
+ state within ``timeout`` seconds.
142
+ ExecutionError: If the execution failed and
143
+ ``raise_on_failure=True``.
144
+ """
145
+ if self.is_done():
146
+ if self.status == "failed" and raise_on_failure:
147
+ raise ExecutionError(
148
+ f"Execution {self.id} failed: {self.error_message}"
149
+ )
150
+ return self
151
+
152
+ deadline = time.monotonic() + float(timeout)
153
+ while True:
154
+ time.sleep(poll_interval)
155
+ self.refresh()
156
+ if self.is_done():
157
+ break
158
+ if time.monotonic() > deadline:
159
+ raise TimeoutError(
160
+ f"Execution {self.id} did not complete within {timeout}s "
161
+ f"(last status: {self.status})"
162
+ )
163
+
164
+ if self.status == "failed" and raise_on_failure:
165
+ raise ExecutionError(
166
+ f"Execution {self.id} failed: {self.error_message}"
167
+ )
168
+ return self
@@ -0,0 +1,32 @@
1
+ """Feedback Loop system for improving evaluations over time.
2
+
3
+ Store developer feedback on metric results, retrieve similar past feedback
4
+ as few-shot examples for LLM judges, and calibrate thresholds statistically.
5
+ """
6
+
7
+ from .types import FeedbackEntry, CalibrationProfile, FeedbackStats
8
+ from .store import FeedbackStore, InMemoryFeedbackStore
9
+ from .collector import FeedbackCollector
10
+ from .retriever import FeedbackRetriever
11
+ from .calibrator import ThresholdCalibrator
12
+ from .hooks import configure_feedback, get_default_store
13
+
14
+ # ChromaFeedbackStore requires chromadb — import conditionally
15
+ try:
16
+ from .store import ChromaFeedbackStore
17
+ except Exception:
18
+ pass
19
+
20
+ __all__ = [
21
+ "FeedbackEntry",
22
+ "CalibrationProfile",
23
+ "FeedbackStats",
24
+ "FeedbackStore",
25
+ "InMemoryFeedbackStore",
26
+ "ChromaFeedbackStore",
27
+ "FeedbackCollector",
28
+ "FeedbackRetriever",
29
+ "ThresholdCalibrator",
30
+ "configure_feedback",
31
+ "get_default_store",
32
+ ]
@@ -0,0 +1,160 @@
1
+ """Statistical threshold calibration based on feedback.
2
+
3
+ Optimizes pass/fail thresholds by computing confusion matrices against
4
+ developer-provided correct labels across a range of threshold values.
5
+ """
6
+
7
+ import logging
8
+ import math
9
+ from typing import List, Tuple
10
+
11
+ from .store import FeedbackStore
12
+ from .types import CalibrationProfile, FeedbackEntry
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ class ThresholdCalibrator:
18
+ """Optimizes pass/fail thresholds based on accumulated feedback.
19
+
20
+ For each candidate threshold, computes a confusion matrix against
21
+ the developer's correct_passed labels, and selects the threshold
22
+ that maximizes agreement (accuracy) or F1 score.
23
+
24
+ Args:
25
+ store: FeedbackStore containing feedback entries.
26
+ optimize_for: Metric to optimize. "accuracy" (default) or "f1".
27
+ """
28
+
29
+ def __init__(
30
+ self,
31
+ store: FeedbackStore,
32
+ optimize_for: str = "accuracy",
33
+ ):
34
+ self.store = store
35
+ self.optimize_for = optimize_for
36
+
37
+ def calibrate(
38
+ self,
39
+ metric_name: str,
40
+ threshold_range: Tuple[float, float] = (0.3, 0.9),
41
+ steps: int = 13,
42
+ ) -> CalibrationProfile:
43
+ """Find the optimal pass/fail threshold for a metric.
44
+
45
+ Args:
46
+ metric_name: The metric to calibrate.
47
+ threshold_range: (min_threshold, max_threshold) to search.
48
+ steps: Number of evenly-spaced thresholds to try.
49
+
50
+ Returns:
51
+ CalibrationProfile with the optimal threshold and stats.
52
+
53
+ Raises:
54
+ ValueError: If insufficient feedback (< 5 entries with corrections).
55
+ """
56
+ entries = self.store.get_by_metric(metric_name)
57
+
58
+ # Filter to entries with both a correct label and a score
59
+ usable = [
60
+ e for e in entries
61
+ if e.correct_score is not None
62
+ and e.original_score is not None
63
+ ]
64
+
65
+ if len(usable) < 5:
66
+ raise ValueError(
67
+ f"Need at least 5 feedback entries with corrections to calibrate "
68
+ f"'{metric_name}', but only {len(usable)} found. Submit more feedback first."
69
+ )
70
+
71
+ # Derive correct_passed if not explicitly set
72
+ for e in usable:
73
+ if e.correct_passed is None:
74
+ e.correct_passed = e.correct_score >= 0.5
75
+
76
+ # Search over threshold space
77
+ min_t, max_t = threshold_range
78
+ best_score = -1.0
79
+ best_threshold = 0.5
80
+ best_matrix = (0, 0, 0, 0)
81
+
82
+ for i in range(steps):
83
+ t = min_t + (max_t - min_t) * i / (steps - 1) if steps > 1 else (min_t + max_t) / 2
84
+ tp, fp, tn, fn = self._confusion_matrix(usable, t)
85
+
86
+ if self.optimize_for == "f1":
87
+ score = self._f1(tp, fp, fn)
88
+ else:
89
+ total = tp + fp + tn + fn
90
+ score = (tp + tn) / total if total > 0 else 0.0
91
+
92
+ if score > best_score:
93
+ best_score = score
94
+ best_threshold = t
95
+ best_matrix = (tp, fp, tn, fn)
96
+
97
+ # Compute score statistics
98
+ scores = [e.correct_score for e in usable if e.correct_score is not None]
99
+ mean = sum(scores) / len(scores) if scores else 0.0
100
+ variance = sum((s - mean) ** 2 for s in scores) / len(scores) if scores else 0.0
101
+
102
+ tp, fp, tn, fn = best_matrix
103
+
104
+ profile = CalibrationProfile(
105
+ eval_name=metric_name,
106
+ optimal_threshold=round(best_threshold, 3),
107
+ sample_size=len(usable),
108
+ accuracy_at_threshold=best_score,
109
+ score_mean=round(mean, 4),
110
+ score_std=round(math.sqrt(variance), 4),
111
+ true_positives=tp,
112
+ false_positives=fp,
113
+ true_negatives=tn,
114
+ false_negatives=fn,
115
+ )
116
+
117
+ logger.info(
118
+ f"Calibrated '{metric_name}': threshold={profile.optimal_threshold} "
119
+ f"accuracy={profile.accuracy_at_threshold:.1%} (n={profile.sample_size})"
120
+ )
121
+
122
+ return profile
123
+
124
+ @staticmethod
125
+ def _confusion_matrix(
126
+ entries: List[FeedbackEntry],
127
+ threshold: float,
128
+ ) -> Tuple[int, int, int, int]:
129
+ """Compute confusion matrix at a given threshold.
130
+
131
+ Predicted positive = original_score >= threshold
132
+ Actual positive = correct_passed is True
133
+
134
+ Returns:
135
+ (true_positives, false_positives, true_negatives, false_negatives)
136
+ """
137
+ tp = fp = tn = fn = 0
138
+ for e in entries:
139
+ predicted_pass = e.original_score >= threshold
140
+ actual_pass = e.correct_passed
141
+
142
+ if predicted_pass and actual_pass:
143
+ tp += 1
144
+ elif predicted_pass and not actual_pass:
145
+ fp += 1
146
+ elif not predicted_pass and not actual_pass:
147
+ tn += 1
148
+ else:
149
+ fn += 1
150
+
151
+ return tp, fp, tn, fn
152
+
153
+ @staticmethod
154
+ def _f1(tp: int, fp: int, fn: int) -> float:
155
+ """Compute F1 score from confusion matrix components."""
156
+ precision = tp / (tp + fp) if (tp + fp) > 0 else 0.0
157
+ recall = tp / (tp + fn) if (tp + fn) > 0 else 0.0
158
+ if precision + recall == 0:
159
+ return 0.0
160
+ return 2 * precision * recall / (precision + recall)
@@ -0,0 +1,214 @@
1
+ """User-facing API for the feedback loop system.
2
+
3
+ Provides a clean interface for submitting feedback, querying statistics,
4
+ calibrating thresholds, and creating retrievers.
5
+ """
6
+
7
+ import logging
8
+ from typing import Any, Dict, List, Optional
9
+
10
+ from ..core.result import EvalResult
11
+ from .store import FeedbackStore
12
+ from .types import FeedbackEntry, FeedbackStats
13
+ from .retriever import FeedbackRetriever
14
+ from .calibrator import ThresholdCalibrator
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+
19
+ class FeedbackCollector:
20
+ """Main user-facing class for the feedback loop system.
21
+
22
+ Provides a clean API for:
23
+ - Submitting feedback on metric results
24
+ - Retrieving statistics on accumulated feedback
25
+ - Calibrating thresholds based on feedback
26
+ - Creating a retriever for pipeline integration
27
+
28
+ Usage:
29
+ from fi.evals.feedback import FeedbackCollector, InMemoryFeedbackStore
30
+
31
+ store = InMemoryFeedbackStore() # or ChromaFeedbackStore()
32
+ feedback = FeedbackCollector(store)
33
+
34
+ # After running a metric that gave wrong results:
35
+ result = run_metric("faithfulness", output="...", context="...")
36
+
37
+ # Submit correction
38
+ feedback.submit(
39
+ result,
40
+ inputs={"output": "...", "context": "..."},
41
+ correct_score=0.9,
42
+ correct_reason="The response IS faithful via semantic equivalence.",
43
+ )
44
+
45
+ # Later, get a retriever for pipeline integration
46
+ retriever = feedback.get_retriever()
47
+
48
+ Args:
49
+ store: The FeedbackStore backend.
50
+ """
51
+
52
+ def __init__(self, store: FeedbackStore):
53
+ self.store = store
54
+
55
+ def submit(
56
+ self,
57
+ result: EvalResult,
58
+ *,
59
+ inputs: Dict[str, Any],
60
+ correct_score: Optional[float] = None,
61
+ correct_passed: Optional[bool] = None,
62
+ correct_reason: str = "",
63
+ tags: Optional[List[str]] = None,
64
+ metadata: Optional[Dict[str, Any]] = None,
65
+ ) -> FeedbackEntry:
66
+ """Submit feedback on a metric result.
67
+
68
+ Args:
69
+ result: The EvalResult that needs correction.
70
+ inputs: The original inputs that were used.
71
+ correct_score: What the score SHOULD have been (0.0-1.0).
72
+ correct_passed: What the pass/fail SHOULD have been.
73
+ correct_reason: Why the original result was wrong.
74
+ tags: Optional tags for organizing feedback.
75
+ metadata: Optional metadata dict.
76
+
77
+ Returns:
78
+ The stored FeedbackEntry.
79
+
80
+ Raises:
81
+ ValueError: If neither correct_score nor correct_reason is provided.
82
+ """
83
+ if correct_score is None and not correct_reason:
84
+ raise ValueError(
85
+ "Feedback must include at least one of: correct_score, correct_reason. "
86
+ "If the result was correct, use confirm() instead."
87
+ )
88
+
89
+ entry = FeedbackEntry(
90
+ eval_name=result.eval_name,
91
+ inputs=inputs,
92
+ original_score=result.score,
93
+ original_reason=result.reason,
94
+ original_passed=result.passed,
95
+ correct_score=correct_score,
96
+ correct_passed=correct_passed,
97
+ correct_reason=correct_reason,
98
+ tags=tags or [],
99
+ metadata=metadata or {},
100
+ )
101
+
102
+ self.store.add(entry)
103
+ logger.info(
104
+ f"Feedback submitted for '{result.eval_name}': "
105
+ f"original={result.score} -> corrected={correct_score}"
106
+ )
107
+ return entry
108
+
109
+ def confirm(
110
+ self,
111
+ result: EvalResult,
112
+ *,
113
+ inputs: Dict[str, Any],
114
+ reason: str = "",
115
+ ) -> FeedbackEntry:
116
+ """Confirm that a metric result was correct.
117
+
118
+ Records that the system got it right, which helps calibration
119
+ accuracy calculations. These entries are stored but NOT injected
120
+ as few-shot examples (since they don't correct anything).
121
+
122
+ Args:
123
+ result: The correct EvalResult.
124
+ inputs: The original inputs.
125
+ reason: Optional note on why this was correct.
126
+
127
+ Returns:
128
+ The stored FeedbackEntry.
129
+ """
130
+ entry = FeedbackEntry(
131
+ eval_name=result.eval_name,
132
+ inputs=inputs,
133
+ original_score=result.score,
134
+ original_reason=result.reason,
135
+ original_passed=result.passed,
136
+ correct_score=result.score, # Same as original = confirmed correct
137
+ correct_passed=result.passed,
138
+ correct_reason=reason or "Confirmed correct by developer.",
139
+ tags=["confirmed"],
140
+ )
141
+ self.store.add(entry)
142
+ return entry
143
+
144
+ def stats(self, metric_name: str) -> FeedbackStats:
145
+ """Get aggregate statistics for feedback on a metric.
146
+
147
+ Args:
148
+ metric_name: The metric to get stats for.
149
+
150
+ Returns:
151
+ FeedbackStats with counts and agreement rates.
152
+ """
153
+ entries = self.store.get_by_metric(metric_name)
154
+
155
+ if not entries:
156
+ return FeedbackStats(eval_name=metric_name)
157
+
158
+ total = len(entries)
159
+ agreements = 0
160
+ score_deltas = []
161
+
162
+ for e in entries:
163
+ if e.correct_score is not None and e.original_score is not None:
164
+ delta = e.correct_score - e.original_score
165
+ score_deltas.append(delta)
166
+ # Agreement = within 0.1 of each other
167
+ if abs(delta) < 0.1:
168
+ agreements += 1
169
+
170
+ # Score distribution in 0.1 buckets
171
+ distribution: Dict[str, int] = {}
172
+ for e in entries:
173
+ score = e.correct_score if e.correct_score is not None else e.original_score
174
+ if score is not None:
175
+ bucket = f"{int(score * 10) / 10:.1f}"
176
+ distribution[bucket] = distribution.get(bucket, 0) + 1
177
+
178
+ return FeedbackStats(
179
+ eval_name=metric_name,
180
+ total_entries=total,
181
+ agreement_rate=agreements / total if total > 0 else 0.0,
182
+ avg_score_delta=sum(score_deltas) / len(score_deltas) if score_deltas else 0.0,
183
+ score_distribution=distribution,
184
+ )
185
+
186
+ def get_retriever(self, max_examples: int = 3) -> FeedbackRetriever:
187
+ """Create a FeedbackRetriever wired to this collector's store.
188
+
189
+ Args:
190
+ max_examples: Max few-shot examples to retrieve per query.
191
+
192
+ Returns:
193
+ A FeedbackRetriever instance.
194
+ """
195
+ return FeedbackRetriever(store=self.store, max_examples=max_examples)
196
+
197
+ def calibrate(
198
+ self,
199
+ metric_name: str,
200
+ threshold_range: tuple = (0.3, 0.9),
201
+ steps: int = 13,
202
+ ):
203
+ """Run threshold calibration for a metric.
204
+
205
+ Args:
206
+ metric_name: Metric to calibrate.
207
+ threshold_range: (min, max) thresholds to search.
208
+ steps: Number of threshold steps to try.
209
+
210
+ Returns:
211
+ CalibrationProfile with optimal threshold.
212
+ """
213
+ calibrator = ThresholdCalibrator(self.store)
214
+ return calibrator.calibrate(metric_name, threshold_range, steps)
@@ -0,0 +1,81 @@
1
+ """Integration hooks for wiring feedback into the pipeline.
2
+
3
+ These functions are called from the augmentation flow when a feedback_store
4
+ is provided. Kept in a separate module to avoid circular imports.
5
+ """
6
+
7
+ import logging
8
+ from typing import Any, Dict, Optional
9
+
10
+ from .store import FeedbackStore
11
+ from .retriever import FeedbackRetriever
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+ # Module-level default store (set via configure_feedback)
16
+ _default_store: Optional[FeedbackStore] = None
17
+ _default_max_examples: int = 3
18
+
19
+
20
+ def configure_feedback(
21
+ store: FeedbackStore,
22
+ max_examples: int = 3,
23
+ ) -> None:
24
+ """Set a global default feedback store for all augmented metric runs.
25
+
26
+ After calling this, all augmented runs will automatically retrieve
27
+ feedback examples -- no need to pass feedback_store= every time.
28
+
29
+ Args:
30
+ store: The FeedbackStore to use globally.
31
+ max_examples: Max few-shot examples per query.
32
+
33
+ Usage:
34
+ from fi.evals.feedback import ChromaFeedbackStore, configure_feedback
35
+
36
+ store = ChromaFeedbackStore()
37
+ configure_feedback(store)
38
+
39
+ # Now all augmented runs automatically use feedback
40
+ result = run_metric("faithfulness", ..., augment=True, model="gemini/...")
41
+ """
42
+ global _default_store, _default_max_examples
43
+ _default_store = store
44
+ _default_max_examples = max_examples
45
+ logger.info(f"Feedback configured globally (max_examples={max_examples})")
46
+
47
+
48
+ def get_default_store() -> Optional[FeedbackStore]:
49
+ """Get the globally configured feedback store, if any."""
50
+ return _default_store
51
+
52
+
53
+ def retrieve_feedback_config(
54
+ metric_name: str,
55
+ inputs: Dict[str, Any],
56
+ store: Optional[FeedbackStore] = None,
57
+ config: Optional[Dict[str, Any]] = None,
58
+ max_examples: Optional[int] = None,
59
+ ) -> Dict[str, Any]:
60
+ """Retrieve feedback examples and inject into config dict.
61
+
62
+ Called from the augmentation flow. Can also be called directly.
63
+
64
+ Args:
65
+ metric_name: Metric being run.
66
+ inputs: Current inputs.
67
+ store: Explicit store override. Falls back to global default.
68
+ config: Existing config dict to merge into.
69
+ max_examples: Override for max examples.
70
+
71
+ Returns:
72
+ Config dict with few_shot_examples populated (or unchanged if
73
+ no store is configured / no feedback found).
74
+ """
75
+ effective_store = store or _default_store
76
+ if effective_store is None:
77
+ return dict(config or {})
78
+
79
+ n = max_examples or _default_max_examples
80
+ retriever = FeedbackRetriever(store=effective_store, max_examples=n)
81
+ return retriever.inject_into_config(metric_name, inputs, config)