agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,344 @@
1
+ """
2
+ Source Attribution Metric.
3
+
4
+ Evaluates citation quality in RAG responses - whether the response
5
+ properly cites its sources and whether citations are accurate.
6
+ """
7
+
8
+ import re
9
+ from typing import Any, Dict, List, Optional
10
+
11
+ from ...base_metric import BaseMetric
12
+ from ..types import SourceAttributionInput
13
+ from ..utils import (
14
+ extract_claims,
15
+ check_claim_supported,
16
+ compute_text_similarity,
17
+ split_into_sentences,
18
+ )
19
+
20
+
21
+ class SourceAttribution(BaseMetric[SourceAttributionInput]):
22
+ """
23
+ Evaluates citation quality in RAG responses.
24
+
25
+ Checks:
26
+ 1. Are claims properly cited?
27
+ 2. Are citations accurate (do they support the claim)?
28
+ 3. Is citation coverage complete?
29
+
30
+ Supports multiple citation formats:
31
+ - Bracketed: [1], [2], etc.
32
+ - Inline: (Source A), (Document 1)
33
+ - Footnote: ¹, ², etc.
34
+
35
+ Score: 0.0 (poor attribution) to 1.0 (excellent attribution)
36
+
37
+ Example:
38
+ >>> attribution = SourceAttribution()
39
+ >>> result = attribution.evaluate([{
40
+ ... "response": "Paris is the capital of France [1].",
41
+ ... "contexts": ["Paris is the capital and largest city of France."],
42
+ ... "citation_format": "bracketed"
43
+ ... }])
44
+ """
45
+
46
+ @property
47
+ def metric_name(self) -> str:
48
+ return "source_attribution"
49
+
50
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
51
+ super().__init__(config)
52
+ self.coverage_weight = self.config.get("coverage_weight", 0.5)
53
+ self.accuracy_weight = self.config.get("accuracy_weight", 0.5)
54
+ self.verification_threshold = self.config.get("verification_threshold", 0.5)
55
+
56
+ def compute_one(self, inputs: SourceAttributionInput) -> Dict[str, Any]:
57
+ response = inputs.response
58
+ contexts = inputs.contexts
59
+ citation_format = inputs.citation_format
60
+ require_citations = inputs.require_citations
61
+
62
+ if not response or not response.strip():
63
+ return {"output": 0.0, "reason": "Empty response"}
64
+
65
+ if not contexts:
66
+ return {"output": 0.0, "reason": "No contexts provided"}
67
+
68
+ # 1. Extract citations from response
69
+ citations = self._extract_citations(response, citation_format)
70
+
71
+ # 2. Extract claims from response
72
+ claims = extract_claims(response)
73
+
74
+ # Handle case with no citations
75
+ if not citations:
76
+ if require_citations and claims:
77
+ return {
78
+ "output": 0.0,
79
+ "reason": "No citations found but citations required",
80
+ "total_claims": len(claims),
81
+ "citations_found": 0,
82
+ }
83
+ else:
84
+ # No citations required or no claims to cite
85
+ return {
86
+ "output": 1.0,
87
+ "reason": "No citations needed",
88
+ "total_claims": len(claims),
89
+ }
90
+
91
+ # 3. Check citation coverage
92
+ coverage_result = self._check_coverage(claims, citations, response)
93
+
94
+ # 4. Verify citation accuracy
95
+ accuracy_result = self._verify_accuracy(citations, contexts)
96
+
97
+ # Combined score
98
+ final_score = (
99
+ self.coverage_weight * coverage_result["coverage"] +
100
+ self.accuracy_weight * accuracy_result["accuracy"]
101
+ )
102
+
103
+ return {
104
+ "output": round(final_score, 4),
105
+ "reason": f"Coverage: {coverage_result['coverage']:.0%}, Accuracy: {accuracy_result['accuracy']:.0%}",
106
+ "citation_coverage": round(coverage_result["coverage"], 4),
107
+ "citation_accuracy": round(accuracy_result["accuracy"], 4),
108
+ "total_claims": len(claims),
109
+ "cited_claims": coverage_result["cited_claims"],
110
+ "total_citations": len(citations),
111
+ "accurate_citations": accuracy_result["accurate_count"],
112
+ "uncited_claims": coverage_result["uncited_claims"][:5],
113
+ "inaccurate_citations": accuracy_result["inaccurate"][:5],
114
+ }
115
+
116
+ def _extract_citations(
117
+ self, text: str, format: str
118
+ ) -> List[Dict[str, Any]]:
119
+ """Extract citations based on format."""
120
+ citations = []
121
+
122
+ if format == "bracketed":
123
+ # Match [1], [2], [1,2], [1-3], etc.
124
+ pattern = r"\[(\d+(?:[,\-]\d+)*)\]"
125
+ for match in re.finditer(pattern, text):
126
+ citation_refs = match.group(1)
127
+ # Parse reference numbers
128
+ source_indices = self._parse_citation_refs(citation_refs)
129
+
130
+ # Get surrounding context as the claim
131
+ start = max(0, match.start() - 300)
132
+ claim_text = text[start:match.start()]
133
+ # Get the last sentence before citation
134
+ sentences = split_into_sentences(claim_text)
135
+ claim_text = sentences[-1] if sentences else claim_text[-200:]
136
+
137
+ citations.append({
138
+ "source_indices": source_indices,
139
+ "claim_text": claim_text.strip(),
140
+ "position": match.start(),
141
+ "raw": match.group(0),
142
+ })
143
+
144
+ elif format == "inline":
145
+ # Match (Source 1), (Document A), etc.
146
+ pattern = r"\((?:Source|Document|Doc|Ref)\.?\s*(\d+|[A-Z])\)"
147
+ for match in re.finditer(pattern, text, re.IGNORECASE):
148
+ ref = match.group(1)
149
+ source_idx = int(ref) - 1 if ref.isdigit() else ord(ref.upper()) - ord('A')
150
+
151
+ claim_text = text[max(0, match.start()-300):match.start()]
152
+ sentences = split_into_sentences(claim_text)
153
+ claim_text = sentences[-1] if sentences else claim_text[-200:]
154
+
155
+ citations.append({
156
+ "source_indices": [source_idx],
157
+ "claim_text": claim_text.strip(),
158
+ "position": match.start(),
159
+ "raw": match.group(0),
160
+ })
161
+
162
+ elif format == "footnote":
163
+ # Match superscript numbers: ¹, ², ³ or ^1, ^2
164
+ pattern = r"[¹²³⁴⁵⁶⁷⁸⁹⁰]+|\^(\d+)"
165
+ superscript_map = {"¹": 1, "²": 2, "³": 3, "⁴": 4, "⁵": 5,
166
+ "⁶": 6, "⁷": 7, "⁸": 8, "⁹": 9, "⁰": 0}
167
+
168
+ for match in re.finditer(pattern, text):
169
+ if match.group(1): # ^1 format
170
+ source_idx = int(match.group(1)) - 1
171
+ else: # Superscript format
172
+ # Convert superscript to number
173
+ num_str = "".join(
174
+ str(superscript_map.get(c, ""))
175
+ for c in match.group(0)
176
+ )
177
+ source_idx = int(num_str) - 1 if num_str else 0
178
+
179
+ claim_text = text[max(0, match.start()-300):match.start()]
180
+ sentences = split_into_sentences(claim_text)
181
+ claim_text = sentences[-1] if sentences else claim_text[-200:]
182
+
183
+ citations.append({
184
+ "source_indices": [source_idx],
185
+ "claim_text": claim_text.strip(),
186
+ "position": match.start(),
187
+ "raw": match.group(0),
188
+ })
189
+
190
+ return citations
191
+
192
+ def _parse_citation_refs(self, refs: str) -> List[int]:
193
+ """Parse citation reference string like '1,2' or '1-3'."""
194
+ indices = []
195
+
196
+ parts = refs.split(",")
197
+ for part in parts:
198
+ if "-" in part:
199
+ # Range: 1-3 -> [0, 1, 2]
200
+ start, end = part.split("-")
201
+ indices.extend(range(int(start) - 1, int(end)))
202
+ else:
203
+ indices.append(int(part) - 1)
204
+
205
+ return indices
206
+
207
+ def _check_coverage(
208
+ self, claims: List[str], citations: List[Dict], response: str
209
+ ) -> Dict[str, Any]:
210
+ """Check what proportion of claims have citations."""
211
+ if not claims:
212
+ return {"coverage": 1.0, "cited_claims": 0, "uncited_claims": []}
213
+
214
+ cited_claims = 0
215
+ uncited_claims = []
216
+
217
+ for claim in claims:
218
+ # Check if this claim has an associated citation
219
+ has_citation = self._claim_has_citation(claim, citations, response)
220
+
221
+ if has_citation:
222
+ cited_claims += 1
223
+ else:
224
+ uncited_claims.append(claim[:80] + "..." if len(claim) > 80 else claim)
225
+
226
+ coverage = cited_claims / len(claims)
227
+
228
+ return {
229
+ "coverage": coverage,
230
+ "cited_claims": cited_claims,
231
+ "uncited_claims": uncited_claims,
232
+ }
233
+
234
+ def _claim_has_citation(
235
+ self, claim: str, citations: List[Dict], response: str
236
+ ) -> bool:
237
+ """Check if a claim has an associated citation."""
238
+ claim_lower = claim.lower()
239
+
240
+ # Find claim position in response
241
+ claim_pos = response.lower().find(claim_lower[:50])
242
+ if claim_pos == -1:
243
+ # Try fuzzy match
244
+ for i, citation in enumerate(citations):
245
+ cited_claim = citation.get("claim_text", "").lower()
246
+ if compute_text_similarity(claim_lower, cited_claim) > 0.6:
247
+ return True
248
+ return False
249
+
250
+ # Check if there's a citation near this claim
251
+ claim_end = claim_pos + len(claim)
252
+
253
+ for citation in citations:
254
+ cite_pos = citation["position"]
255
+ # Citation should be within 50 chars after the claim
256
+ if claim_pos <= cite_pos <= claim_end + 50:
257
+ return True
258
+
259
+ return False
260
+
261
+ def _verify_accuracy(
262
+ self, citations: List[Dict], contexts: List[str]
263
+ ) -> Dict[str, Any]:
264
+ """Verify that citations accurately point to supporting sources."""
265
+ if not citations:
266
+ return {"accuracy": 1.0, "accurate_count": 0, "inaccurate": []}
267
+
268
+ accurate = 0
269
+ inaccurate = []
270
+
271
+ for citation in citations:
272
+ source_indices = citation.get("source_indices", [])
273
+ claim_text = citation.get("claim_text", "")
274
+
275
+ if not claim_text:
276
+ continue
277
+
278
+ # Check if any cited source supports the claim
279
+ is_accurate = False
280
+ for idx in source_indices:
281
+ if 0 <= idx < len(contexts):
282
+ source = contexts[idx]
283
+ is_supported, score, _ = check_claim_supported(
284
+ claim_text, [source], self.verification_threshold
285
+ )
286
+ if is_supported:
287
+ is_accurate = True
288
+ break
289
+
290
+ if is_accurate:
291
+ accurate += 1
292
+ else:
293
+ inaccurate.append({
294
+ "claim": claim_text[:80],
295
+ "cited_sources": source_indices,
296
+ })
297
+
298
+ accuracy = accurate / len(citations) if citations else 1.0
299
+
300
+ return {
301
+ "accuracy": accuracy,
302
+ "accurate_count": accurate,
303
+ "inaccurate": inaccurate,
304
+ }
305
+
306
+
307
+ class CitationPresence(BaseMetric[SourceAttributionInput]):
308
+ """
309
+ Simple metric to check if citations are present.
310
+
311
+ Useful as a quick check before detailed attribution analysis.
312
+
313
+ Score: 0.0 (no citations) to 1.0 (citations present)
314
+ """
315
+
316
+ @property
317
+ def metric_name(self) -> str:
318
+ return "citation_presence"
319
+
320
+ def compute_one(self, inputs: SourceAttributionInput) -> Dict[str, Any]:
321
+ response = inputs.response
322
+
323
+ # Check for any citation patterns
324
+ patterns = [
325
+ r"\[\d+\]", # [1]
326
+ r"\(\d+\)", # (1)
327
+ r"\((?:Source|Document|Ref)\s*\d+\)", # (Source 1)
328
+ r"[¹²³⁴⁵⁶⁷⁸⁹]", # Superscripts
329
+ r"\^\d+", # ^1
330
+ ]
331
+
332
+ citations_found = []
333
+ for pattern in patterns:
334
+ matches = re.findall(pattern, response, re.IGNORECASE)
335
+ citations_found.extend(matches)
336
+
337
+ has_citations = len(citations_found) > 0
338
+
339
+ return {
340
+ "output": 1.0 if has_citations else 0.0,
341
+ "reason": f"Found {len(citations_found)} citations" if has_citations else "No citations found",
342
+ "citations_found": citations_found[:10],
343
+ "citation_count": len(citations_found),
344
+ }
@@ -0,0 +1,17 @@
1
+ """
2
+ RAG Generation Metrics.
3
+
4
+ Metrics for evaluating the generation component of RAG systems.
5
+ """
6
+
7
+ from .answer_relevancy import AnswerRelevancy
8
+ from .context_utilization import ContextUtilization
9
+ from .groundedness import Groundedness
10
+ from .faithfulness import RAGFaithfulness
11
+
12
+ __all__ = [
13
+ "AnswerRelevancy",
14
+ "ContextUtilization",
15
+ "Groundedness",
16
+ "RAGFaithfulness",
17
+ ]
@@ -0,0 +1,176 @@
1
+ """
2
+ Answer Relevancy Metric.
3
+
4
+ Measures how well the generated response addresses the original query.
5
+ """
6
+
7
+ from typing import Any, Dict, Optional
8
+
9
+ from ...base_metric import BaseMetric
10
+ from ..types import AnswerRelevancyInput
11
+ from ..utils import (
12
+ extract_keywords,
13
+ compute_semantic_similarity,
14
+ )
15
+
16
+
17
+ class AnswerRelevancy(BaseMetric[AnswerRelevancyInput]):
18
+ """
19
+ Measures how well the response addresses the query.
20
+
21
+ Combines multiple signals:
22
+ - Keyword coverage (query keywords in response)
23
+ - Semantic similarity (embedding-based)
24
+ - Direct answer indicators
25
+
26
+ Penalizes:
27
+ - Off-topic responses
28
+ - Incomplete answers
29
+ - Over-general responses
30
+
31
+ Score: 0.0 (irrelevant) to 1.0 (highly relevant)
32
+
33
+ Example:
34
+ >>> relevancy = AnswerRelevancy()
35
+ >>> result = relevancy.evaluate([{
36
+ ... "query": "What is the capital of France?",
37
+ ... "response": "The capital of France is Paris."
38
+ ... }])
39
+ """
40
+
41
+ @property
42
+ def metric_name(self) -> str:
43
+ return "answer_relevancy"
44
+
45
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
46
+ super().__init__(config)
47
+ self.keyword_weight = self.config.get("keyword_weight", 0.3)
48
+ self.semantic_weight = self.config.get("semantic_weight", 0.5)
49
+ self.structure_weight = self.config.get("structure_weight", 0.2)
50
+
51
+ def compute_one(self, inputs: AnswerRelevancyInput) -> Dict[str, Any]:
52
+ query = inputs.query
53
+ response = inputs.response
54
+
55
+ if not response or not response.strip():
56
+ return {
57
+ "output": 0.0,
58
+ "reason": "Empty response",
59
+ }
60
+
61
+ if not query or not query.strip():
62
+ return {
63
+ "output": 1.0,
64
+ "reason": "No query to evaluate against",
65
+ }
66
+
67
+ # 1. Keyword coverage
68
+ query_keywords = extract_keywords(query)
69
+ response_keywords = extract_keywords(response)
70
+
71
+ if query_keywords:
72
+ overlap = len(query_keywords & response_keywords)
73
+ keyword_coverage = overlap / len(query_keywords)
74
+ else:
75
+ keyword_coverage = 0.5 # Neutral if no keywords
76
+
77
+ # 2. Semantic similarity
78
+ semantic_sim = compute_semantic_similarity(query, response)
79
+
80
+ # 3. Structural relevancy indicators
81
+ structure_score = self._check_structure(query, response)
82
+
83
+ # 4. Check for refusal or non-answer patterns
84
+ refusal_penalty = self._check_refusal(response)
85
+
86
+ # Combine scores
87
+ base_score = (
88
+ self.keyword_weight * keyword_coverage +
89
+ self.semantic_weight * semantic_sim +
90
+ self.structure_weight * structure_score
91
+ )
92
+
93
+ # Apply refusal penalty
94
+ final_score = base_score * (1.0 - refusal_penalty)
95
+
96
+ return {
97
+ "output": round(final_score, 4),
98
+ "reason": f"Relevancy={final_score:.2f} (keywords={keyword_coverage:.2f}, semantic={semantic_sim:.2f})",
99
+ "keyword_coverage": round(keyword_coverage, 4),
100
+ "semantic_similarity": round(semantic_sim, 4),
101
+ "structure_score": round(structure_score, 4),
102
+ "refusal_penalty": round(refusal_penalty, 4),
103
+ }
104
+
105
+ def _check_structure(self, query: str, response: str) -> float:
106
+ """Check structural indicators of a relevant answer."""
107
+ score = 0.5 # Neutral starting point
108
+ query_lower = query.lower()
109
+ response_lower = response.lower()
110
+
111
+ # Question type detection and answer format checking
112
+ question_patterns = {
113
+ "what is": ["is", "are", "the", "a", "an"],
114
+ "what are": ["are", "include", "consist"],
115
+ "who is": ["is", "was", "name"],
116
+ "who are": ["are", "were", "include"],
117
+ "when": ["in", "on", "at", "during", "year", "date"],
118
+ "where": ["in", "at", "located", "place", "city", "country"],
119
+ "how": ["by", "through", "using", "step", "method"],
120
+ "why": ["because", "since", "due to", "reason", "cause"],
121
+ "how many": ["number", "total", "count"] + [str(i) for i in range(10)],
122
+ "how much": ["amount", "cost", "price", "$", "dollar"],
123
+ }
124
+
125
+ for question_type, answer_indicators in question_patterns.items():
126
+ if question_type in query_lower:
127
+ # Check if response has appropriate indicators
128
+ has_indicator = any(ind in response_lower for ind in answer_indicators)
129
+ if has_indicator:
130
+ score += 0.3
131
+ break
132
+
133
+ # Check for direct answer patterns
134
+ direct_starters = [
135
+ "the answer is",
136
+ "it is",
137
+ "they are",
138
+ "yes,",
139
+ "no,",
140
+ "the",
141
+ ]
142
+ if any(response_lower.startswith(starter) for starter in direct_starters):
143
+ score += 0.2
144
+
145
+ return min(1.0, score)
146
+
147
+ def _check_refusal(self, response: str) -> float:
148
+ """Check for refusal or non-answer patterns."""
149
+ response_lower = response.lower()
150
+
151
+ refusal_patterns = [
152
+ "i cannot",
153
+ "i can't",
154
+ "i'm unable",
155
+ "i don't know",
156
+ "i do not know",
157
+ "i'm not sure",
158
+ "i am not sure",
159
+ "i don't have",
160
+ "i do not have",
161
+ "i cannot provide",
162
+ "unable to answer",
163
+ "don't have information",
164
+ "no information",
165
+ "not available",
166
+ ]
167
+
168
+ for pattern in refusal_patterns:
169
+ if pattern in response_lower:
170
+ return 0.5 # Partial penalty for refusal
171
+
172
+ # Check for extremely short responses
173
+ if len(response.split()) < 3:
174
+ return 0.2 # Small penalty for very short responses
175
+
176
+ return 0.0