agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,253 @@
1
+ """
2
+ Natural Language Inference utilities for hallucination detection.
3
+
4
+ Provides entailment checking between claims and context.
5
+ Uses transformer-based NLI when available (pip install ai-evaluation[nli]),
6
+ falls back to word-overlap heuristic with a warning.
7
+ """
8
+
9
+ import re
10
+ import warnings
11
+ from typing import Tuple, List, Optional
12
+ from enum import Enum
13
+
14
+
15
+ _NLI_MODEL = "cross-encoder/nli-deberta-v3-base"
16
+
17
+ # Try to import transformers
18
+ _NLI_AVAILABLE = False
19
+ try:
20
+ from transformers import pipeline as _hf_pipeline
21
+ _NLI_AVAILABLE = True
22
+ except ImportError:
23
+ pass
24
+
25
+ _nli_pipeline = None
26
+ _heuristic_warning_issued = False
27
+
28
+
29
+ class NLILabel(Enum):
30
+ """NLI classification labels."""
31
+
32
+ ENTAILMENT = "entailment"
33
+ CONTRADICTION = "contradiction"
34
+ NEUTRAL = "neutral"
35
+
36
+
37
+ def _get_nli_pipeline():
38
+ """Lazy-load NLI pipeline. Returns the pipeline or False on failure."""
39
+ global _nli_pipeline
40
+ if _nli_pipeline is not None:
41
+ return _nli_pipeline
42
+
43
+ if not _NLI_AVAILABLE:
44
+ _nli_pipeline = False
45
+ return False
46
+
47
+ try:
48
+ _nli_pipeline = _hf_pipeline(
49
+ "text-classification",
50
+ model=_NLI_MODEL,
51
+ device=-1, # CPU
52
+ )
53
+ except Exception as exc:
54
+ warnings.warn(
55
+ f"Failed to load NLI model '{_NLI_MODEL}': {exc}. "
56
+ "Falling back to word-overlap heuristic. "
57
+ "Install with: pip install ai-evaluation[nli]",
58
+ RuntimeWarning,
59
+ stacklevel=2,
60
+ )
61
+ _nli_pipeline = False
62
+
63
+ return _nli_pipeline
64
+
65
+
66
+ def _warn_heuristic_fallback():
67
+ """Issue a one-time warning that we're using the heuristic fallback."""
68
+ global _heuristic_warning_issued
69
+ if not _heuristic_warning_issued:
70
+ _heuristic_warning_issued = True
71
+ warnings.warn(
72
+ "NLI model not available — using word-overlap heuristic for "
73
+ "hallucination detection. Results will be approximate. "
74
+ "For accurate NLI, install: pip install ai-evaluation[nli]",
75
+ RuntimeWarning,
76
+ stacklevel=3,
77
+ )
78
+
79
+
80
+ _LABEL_MAP = {
81
+ "ENTAILMENT": NLILabel.ENTAILMENT,
82
+ "CONTRADICTION": NLILabel.CONTRADICTION,
83
+ "NEUTRAL": NLILabel.NEUTRAL,
84
+ "entailment": NLILabel.ENTAILMENT,
85
+ "contradiction": NLILabel.CONTRADICTION,
86
+ "neutral": NLILabel.NEUTRAL,
87
+ }
88
+
89
+
90
+ def check_entailment(premise: str, hypothesis: str) -> Tuple[NLILabel, float]:
91
+ """
92
+ Check if premise entails hypothesis using NLI model.
93
+
94
+ Args:
95
+ premise: The source text (context)
96
+ hypothesis: The claim to verify
97
+
98
+ Returns:
99
+ Tuple of (NLI label, confidence score)
100
+ """
101
+ nli = _get_nli_pipeline()
102
+
103
+ if not nli:
104
+ _warn_heuristic_fallback()
105
+ return check_entailment_heuristic(premise, hypothesis)
106
+
107
+ try:
108
+ result = nli(
109
+ {"text": premise, "text_pair": hypothesis},
110
+ truncation=True,
111
+ max_length=512,
112
+ )
113
+ # Dict input returns a single dict, not a list
114
+ entry = result[0] if isinstance(result, list) else result
115
+ label = _LABEL_MAP.get(entry["label"], NLILabel.NEUTRAL)
116
+ score = entry["score"]
117
+ return label, score
118
+ except Exception:
119
+ return check_entailment_heuristic(premise, hypothesis)
120
+
121
+
122
+ # ---------------------------------------------------------------------------
123
+ # Heuristic fallback
124
+ # ---------------------------------------------------------------------------
125
+
126
+ _STOPWORDS = frozenset({
127
+ "the", "a", "an", "is", "are", "was", "were", "be", "been", "being",
128
+ "have", "has", "had", "do", "does", "did", "will", "would", "could",
129
+ "should", "may", "might", "must", "shall", "can", "to", "of", "in",
130
+ "for", "on", "with", "at", "by", "from", "as", "and", "or", "but",
131
+ "if", "that", "this", "it", "its", "they", "their", "he", "she",
132
+ "him", "her", "his", "we", "our", "you", "your",
133
+ })
134
+
135
+ _NEGATIONS = frozenset({
136
+ "not", "n't", "never", "no", "none", "neither", "nor", "cannot",
137
+ })
138
+
139
+
140
+ def _tokenize(text: str) -> set:
141
+ """Tokenize text into content words, stripping punctuation."""
142
+ return set(re.findall(r"\b\w+\b", text.lower()))
143
+
144
+
145
+ def check_entailment_heuristic(
146
+ premise: str, hypothesis: str
147
+ ) -> Tuple[NLILabel, float]:
148
+ """
149
+ Heuristic entailment check using word overlap and similarity.
150
+
151
+ Fallback when NLI model is not available. Uses:
152
+ - Content word overlap for entailment signal
153
+ - Negation asymmetry for contradiction detection
154
+ - Numeric mismatch detection
155
+
156
+ Args:
157
+ premise: The source text (context)
158
+ hypothesis: The claim to verify
159
+
160
+ Returns:
161
+ Tuple of (NLI label, confidence score)
162
+ """
163
+ premise_words = _tokenize(premise)
164
+ hypothesis_words = _tokenize(hypothesis)
165
+
166
+ premise_content = premise_words - _STOPWORDS
167
+ hypothesis_content = hypothesis_words - _STOPWORDS
168
+
169
+ if not hypothesis_content:
170
+ return NLILabel.NEUTRAL, 0.5
171
+
172
+ overlap = len(premise_content & hypothesis_content)
173
+ coverage = overlap / len(hypothesis_content)
174
+
175
+ # Check for negation asymmetry
176
+ premise_has_neg = bool(premise_words & _NEGATIONS)
177
+ hypothesis_has_neg = bool(hypothesis_words & _NEGATIONS)
178
+
179
+ if premise_has_neg != hypothesis_has_neg and coverage > 0.5:
180
+ return NLILabel.CONTRADICTION, 0.6
181
+
182
+ # Check for numeric mismatch — only when high overlap + claim has numbers not in premise
183
+ premise_numbers = set(re.findall(r"\b\d+(?:\.\d+)?\b", premise))
184
+ hypothesis_numbers = set(re.findall(r"\b\d+(?:\.\d+)?\b", hypothesis))
185
+
186
+ if hypothesis_numbers and premise_numbers and coverage > 0.5:
187
+ novel_numbers = hypothesis_numbers - premise_numbers
188
+ if novel_numbers and len(novel_numbers) <= 2:
189
+ return NLILabel.CONTRADICTION, 0.55
190
+
191
+ if coverage >= 0.65:
192
+ return NLILabel.ENTAILMENT, coverage
193
+ elif coverage >= 0.3:
194
+ return NLILabel.NEUTRAL, coverage
195
+ else:
196
+ return NLILabel.NEUTRAL, coverage
197
+
198
+
199
+ def check_contradiction(claim: str, context: str) -> Tuple[bool, float]:
200
+ """
201
+ Check if claim contradicts the context.
202
+
203
+ Args:
204
+ claim: The claim to check
205
+ context: The context to check against
206
+
207
+ Returns:
208
+ Tuple of (is_contradiction, confidence)
209
+ """
210
+ label, score = check_entailment(context, claim)
211
+
212
+ if label == NLILabel.CONTRADICTION:
213
+ return True, score
214
+
215
+ return False, 0.0
216
+
217
+
218
+ def nli_score_for_claim(
219
+ claim: str, contexts: List[str]
220
+ ) -> Tuple[NLILabel, float, Optional[str]]:
221
+ """
222
+ Get the best NLI score for a claim against multiple contexts.
223
+
224
+ Args:
225
+ claim: The claim to verify
226
+ contexts: List of context passages
227
+
228
+ Returns:
229
+ Tuple of (best_label, best_score, best_context_snippet)
230
+ """
231
+ best_label = NLILabel.NEUTRAL
232
+ best_score = 0.0
233
+ best_context = None
234
+
235
+ for ctx in contexts:
236
+ label, score = check_entailment(ctx, claim)
237
+
238
+ if label == NLILabel.CONTRADICTION and score > 0.5:
239
+ # Contradiction takes priority if confident
240
+ snippet = ctx[:200] + "..." if len(ctx) > 200 else ctx
241
+ return label, score, snippet
242
+
243
+ if label == NLILabel.ENTAILMENT and score > best_score:
244
+ best_label = label
245
+ best_score = score
246
+ best_context = ctx[:200] + "..." if len(ctx) > 200 else ctx
247
+ elif label == NLILabel.NEUTRAL and best_label != NLILabel.ENTAILMENT:
248
+ if score > best_score:
249
+ best_label = label
250
+ best_score = score
251
+ best_context = ctx[:200] + "..." if len(ctx) > 200 else ctx
252
+
253
+ return best_label, best_score, best_context
@@ -0,0 +1,106 @@
1
+ """
2
+ Hallucination Sentinel — fast pre-screening before full NLI.
3
+
4
+ Provides rule-based screening to quickly flag responses
5
+ that are likely or unlikely to contain hallucinations,
6
+ avoiding expensive NLI inference for obvious cases.
7
+ """
8
+
9
+ import re
10
+ from typing import Dict, List, Literal, Tuple
11
+
12
+
13
+ RiskLevel = Literal["low", "medium", "high"]
14
+
15
+
16
+ # Patterns that indicate high hallucination risk
17
+ _HIGH_RISK_PATTERNS = [
18
+ r"\baccording to (?:recent|latest|new)\b",
19
+ r"\bstudies (?:show|prove|confirm|suggest)\b",
20
+ r"\bresearch (?:shows|proves|confirms|suggests)\b",
21
+ r"\bstatistics (?:show|indicate|reveal)\b",
22
+ r"\b\d+(?:\.\d+)?%\b", # Specific percentages
23
+ r"\bin \d{4}\b", # Specific years
24
+ r"\bexactly \d+\b", # Exact numbers
25
+ r"\bproven (?:fact|to be)\b",
26
+ r"\bit is (?:well[- ]known|widely accepted|universally agreed)\b",
27
+ ]
28
+
29
+ # Patterns that indicate the response is hedging (lower risk)
30
+ _HEDGE_PATTERNS = [
31
+ r"\bI (?:think|believe|am not sure)\b",
32
+ r"\b(?:might|may|could|possibly|perhaps|probably)\b",
33
+ r"\b(?:it seems|it appears|it looks like)\b",
34
+ r"\bI don't (?:know|have)\b",
35
+ r"\bnot (?:certain|sure|clear)\b",
36
+ ]
37
+
38
+
39
+ class HallucinationSentinel:
40
+ """Fast rule-based screening for hallucination risk."""
41
+
42
+ def __init__(
43
+ self,
44
+ extra_risk_patterns: List[str] = None,
45
+ ):
46
+ self.risk_patterns = _HIGH_RISK_PATTERNS + (extra_risk_patterns or [])
47
+
48
+ def screen(
49
+ self, response: str, context: str
50
+ ) -> Tuple[RiskLevel, Dict]:
51
+ """
52
+ Screen a response for hallucination risk.
53
+
54
+ Args:
55
+ response: The LLM response to screen
56
+ context: The source context
57
+
58
+ Returns:
59
+ Tuple of (risk_level, details dict)
60
+ """
61
+ response_lower = response.lower()
62
+ context_lower = context.lower()
63
+
64
+ details: Dict = {
65
+ "risk_signals": [],
66
+ "hedge_signals": [],
67
+ }
68
+
69
+ # Check risk patterns
70
+ risk_count = 0
71
+ for pattern in self.risk_patterns:
72
+ matches = re.findall(pattern, response_lower, re.IGNORECASE)
73
+ if matches:
74
+ risk_count += len(matches)
75
+ details["risk_signals"].append(pattern)
76
+
77
+ # Check hedging patterns
78
+ hedge_count = 0
79
+ for pattern in _HEDGE_PATTERNS:
80
+ if re.search(pattern, response_lower, re.IGNORECASE):
81
+ hedge_count += 1
82
+ details["hedge_signals"].append(pattern)
83
+
84
+ # Check if claims reference things not in context
85
+ # Simple: response words not found in context
86
+ response_words = set(re.findall(r"\b\w{4,}\b", response_lower))
87
+ context_words = set(re.findall(r"\b\w{4,}\b", context_lower))
88
+ novel_ratio = len(response_words - context_words) / max(len(response_words), 1)
89
+ details["novel_word_ratio"] = round(novel_ratio, 3)
90
+
91
+ # Determine risk level
92
+ if risk_count >= 3 or (risk_count >= 1 and novel_ratio > 0.6):
93
+ risk_level: RiskLevel = "high"
94
+ elif risk_count >= 1 or novel_ratio > 0.5:
95
+ risk_level = "medium"
96
+ else:
97
+ risk_level = "low"
98
+
99
+ # Hedging reduces risk
100
+ if hedge_count >= 2 and risk_level == "high":
101
+ risk_level = "medium"
102
+
103
+ details["risk_count"] = risk_count
104
+ details["hedge_count"] = hedge_count
105
+
106
+ return risk_level, details
@@ -0,0 +1,132 @@
1
+ """
2
+ Types for Hallucination Detection.
3
+
4
+ These types support NLI-based and semantic analysis for
5
+ detecting hallucinations in LLM outputs.
6
+ """
7
+
8
+ from typing import Any, Dict, List, Literal, Optional, Union
9
+ from pydantic import BaseModel, Field
10
+
11
+ from ...types import BaseMetricInput
12
+
13
+
14
+ class Claim(BaseModel):
15
+ """Represents a single claim extracted from text."""
16
+
17
+ text: str = Field(..., description="The claim text")
18
+ source_span: Optional[str] = Field(
19
+ default=None,
20
+ description="Original text span the claim was extracted from"
21
+ )
22
+ confidence: Optional[float] = Field(
23
+ default=None,
24
+ description="Confidence score for claim extraction (0-1)"
25
+ )
26
+
27
+
28
+ class HallucinationInput(BaseMetricInput):
29
+ """
30
+ Input for hallucination detection metrics.
31
+
32
+ Evaluates whether the response contains claims not supported
33
+ by the provided context/source.
34
+ """
35
+
36
+ # The LLM response to check for hallucinations
37
+ response: str = Field(
38
+ ...,
39
+ description="The LLM response to evaluate for hallucinations."
40
+ )
41
+
42
+ # The source/context that the response should be faithful to
43
+ context: Union[str, List[str]] = Field(
44
+ ...,
45
+ description="Source context(s) the response should be grounded in."
46
+ )
47
+
48
+ # Optional: pre-extracted claims from response
49
+ claims: Optional[List[Claim]] = Field(
50
+ default=None,
51
+ description="Pre-extracted claims from response. If not provided, claims are extracted automatically."
52
+ )
53
+
54
+ # Optional: the query that generated the response
55
+ query: Optional[str] = Field(
56
+ default=None,
57
+ description="The original query/question that generated the response."
58
+ )
59
+
60
+
61
+ class ClaimExtractionInput(BaseMetricInput):
62
+ """
63
+ Input for claim extraction from text.
64
+
65
+ Extracts atomic, verifiable claims from a text passage.
66
+ """
67
+
68
+ response: str = Field(
69
+ ...,
70
+ description="The text to extract claims from."
71
+ )
72
+
73
+ # Extraction granularity
74
+ granularity: Literal["sentence", "clause", "atomic"] = Field(
75
+ default="sentence",
76
+ description="Granularity of claim extraction: sentence, clause, or atomic (finest)."
77
+ )
78
+
79
+
80
+ class FactualConsistencyInput(BaseMetricInput):
81
+ """
82
+ Input for factual consistency checking.
83
+
84
+ Evaluates whether claims in the response are consistent with
85
+ known facts or a reference text.
86
+ """
87
+
88
+ response: str = Field(
89
+ ...,
90
+ description="The response to check for factual consistency."
91
+ )
92
+
93
+ # Reference for fact-checking
94
+ reference: Optional[str] = Field(
95
+ default=None,
96
+ description="Reference text containing ground truth facts."
97
+ )
98
+
99
+ # Pre-extracted claims
100
+ claims: Optional[List[Claim]] = Field(
101
+ default=None,
102
+ description="Pre-extracted claims from the response."
103
+ )
104
+
105
+
106
+ class NLIResult(BaseModel):
107
+ """Result of Natural Language Inference classification."""
108
+
109
+ premise: str = Field(..., description="The premise (context)")
110
+ hypothesis: str = Field(..., description="The hypothesis (claim)")
111
+ label: Literal["entailment", "neutral", "contradiction"] = Field(
112
+ ...,
113
+ description="NLI classification label"
114
+ )
115
+ scores: Dict[str, float] = Field(
116
+ default_factory=dict,
117
+ description="Probability scores for each label"
118
+ )
119
+
120
+
121
+ class HallucinationResult(BaseModel):
122
+ """Detailed result of hallucination detection."""
123
+
124
+ score: float = Field(..., description="Overall hallucination score (0=hallucinated, 1=faithful)")
125
+ claims_analyzed: int = Field(..., description="Number of claims analyzed")
126
+ supported_claims: int = Field(..., description="Number of claims supported by context")
127
+ unsupported_claims: int = Field(..., description="Number of unsupported claims")
128
+ contradicted_claims: int = Field(..., description="Number of contradicted claims")
129
+ claim_details: List[Dict[str, Any]] = Field(
130
+ default_factory=list,
131
+ description="Detailed analysis for each claim"
132
+ )
@@ -0,0 +1,85 @@
1
+ import json
2
+ from typing import Any, Dict, List, Optional
3
+
4
+ from ..base_metric import BaseMetric, BaseMetricInputType
5
+
6
+
7
+ class AggregatedMetric(BaseMetric[BaseMetricInputType]):
8
+ """
9
+ Combines multiple metric evaluators into a single aggregated score.
10
+
11
+ This metric assumes all sub-metrics can operate on the same input type.
12
+
13
+ Config:
14
+ - aggregator (str): 'average' or 'weighted_average'.
15
+ - metrics (List[BaseMetricInput]): A list of instantiated metric objects.
16
+ - weights (List[float]): Required if aggregator is 'weighted_average'.
17
+ """
18
+
19
+ SUPPORTED_AGGREGATORS = ["average", "weighted_average"]
20
+
21
+ @property
22
+ def metric_name(self) -> str:
23
+ return "aggregated_metric"
24
+
25
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
26
+ super().__init__(config)
27
+ self.aggregator = self.config.get("aggregator", "average")
28
+ self.metrics: List[BaseMetric] = self.config.get("metrics", [])
29
+ self.weights: List[float] = self.config.get("weights", [])
30
+
31
+ if self.aggregator not in self.SUPPORTED_AGGREGATORS:
32
+ raise ValueError(f"Unsupported aggregator: {self.aggregator}")
33
+ if not self.metrics:
34
+ raise ValueError(
35
+ "AggregatedMetric requires at least one metric in its config."
36
+ )
37
+ if not all(isinstance(m, BaseMetric) for m in self.metrics):
38
+ raise TypeError("All items in 'metrics' must be instances of BaseMetric.")
39
+
40
+ # Explicitly set the input model based on the first sub-metric
41
+ self.input_model = self.metrics[0].input_model
42
+
43
+ if self.aggregator == "weighted_average":
44
+ if not self.weights or len(self.weights) != len(self.metrics):
45
+ raise ValueError(
46
+ "Weights are required for 'weighted_average' and must match the number of metrics."
47
+ )
48
+
49
+ def _normalize_score(self, value: Any) -> float:
50
+ """Converts various score types to a float, clamping between 0 and 1."""
51
+ if isinstance(value, bool):
52
+ return 1.0 if value else 0.0
53
+ try:
54
+ float_value = float(value)
55
+ return max(0.0, min(1.0, float_value))
56
+ except (ValueError, TypeError):
57
+ return 0.0
58
+
59
+ def compute_one(self, inputs: BaseMetricInputType) -> Dict[str, Any]:
60
+ metric_scores = []
61
+ metric_details = {}
62
+
63
+ for metric in self.metrics:
64
+ try:
65
+ result_dict = metric.compute_one(inputs)
66
+ score = self._normalize_score(result_dict.get("output", 0.0))
67
+ except Exception:
68
+ # If a sub-metric fails, record a score of 0.0 for it
69
+ score = 0.0
70
+
71
+ metric_scores.append(score)
72
+ metric_details[metric.metric_name] = score
73
+
74
+ if not metric_scores:
75
+ return {"output": 0.0, "reason": "No metric scores were produced."}
76
+
77
+ if self.aggregator == "average":
78
+ aggregated_score = sum(metric_scores) / len(metric_scores)
79
+ else: # weighted_average
80
+ weighted_sum = sum(w * s for w, s in zip(self.weights, metric_scores))
81
+ total_weight = sum(self.weights)
82
+ aggregated_score = weighted_sum / total_weight if total_weight > 0 else 0.0
83
+
84
+ reason = f"Aggregated score calculated using '{self.aggregator}'. Details: {json.dumps(metric_details)}"
85
+ return {"output": aggregated_score, "reason": reason}
@@ -0,0 +1,87 @@
1
+ import json
2
+ from typing import Any, Dict
3
+ import re
4
+
5
+ # This would ideally be in a separate helper file
6
+ from jsonschema import validate
7
+ from jsonschema.exceptions import ValidationError as JsonSchemaValidationError
8
+
9
+ from ..base_metric import BaseMetric
10
+ from ...types import TextMetricInput, JsonMetricInput
11
+
12
+
13
+ class ContainsJson(BaseMetric[TextMetricInput]):
14
+ """Checks if the response text contains a valid JSON object or array."""
15
+
16
+ @property
17
+ def metric_name(self) -> str:
18
+ return "contains_json"
19
+
20
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
21
+ text = inputs.response.strip()
22
+ # Simple regex to find potential JSON candidates
23
+ json_candidates = re.findall(r"\{.*\}|\[.*\]", text, re.DOTALL)
24
+ for candidate in json_candidates:
25
+ try:
26
+ json.loads(candidate)
27
+ return {
28
+ "output": 1.0,
29
+ "reason": "A valid JSON entity was found in the response.",
30
+ }
31
+ except json.JSONDecodeError:
32
+ continue
33
+ return {"output": 0.0, "reason": "No valid JSON entity found in the response."}
34
+
35
+
36
+ class IsJson(BaseMetric[TextMetricInput]):
37
+ """Checks if the entire response text is a single, valid JSON object."""
38
+
39
+ @property
40
+ def metric_name(self) -> str:
41
+ return "is_json"
42
+
43
+ def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
44
+ try:
45
+ json.loads(inputs.response)
46
+ return {"output": 1.0, "reason": "Response is a valid JSON object."}
47
+ except json.JSONDecodeError as e:
48
+ return {
49
+ "output": 0.0,
50
+ "reason": f"Response is not a valid JSON object: {e}",
51
+ }
52
+
53
+
54
+ class JsonSchema(BaseMetric[JsonMetricInput]):
55
+ """Validates the `response` against a provided JSON schema."""
56
+
57
+ @property
58
+ def metric_name(self) -> str:
59
+ return "json_schema"
60
+
61
+ def compute_one(self, inputs: JsonMetricInput) -> Dict[str, Any]:
62
+ if not inputs.schema:
63
+ raise ValueError("JsonSchema metric requires 'schema' to be provided.")
64
+ try:
65
+ actual_data = (
66
+ json.loads(inputs.response)
67
+ if isinstance(inputs.response, str)
68
+ else inputs.response
69
+ )
70
+ except json.JSONDecodeError as e:
71
+ return {"output": 0.0, "reason": f"Actual JSON is invalid: {e}"}
72
+ try:
73
+ schema_data = (
74
+ json.loads(inputs.schema)
75
+ if isinstance(inputs.schema, str)
76
+ else inputs.schema
77
+ )
78
+ except json.JSONDecodeError as e:
79
+ return {"output": 0.0, "reason": f"Schema JSON is invalid: {e}"}
80
+ try:
81
+ validate(instance=actual_data, schema=schema_data)
82
+ return {"output": 1.0, "reason": "JSON conforms to the schema."}
83
+ except JsonSchemaValidationError as e:
84
+ return {
85
+ "output": 0.0,
86
+ "reason": f"JSON schema validation failed: {e.message}",
87
+ }