agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,298 @@
1
+ """
2
+ Structured Output Score - Composite Metric.
3
+
4
+ Combines multiple structured validation aspects into a single score.
5
+ """
6
+
7
+ from typing import Any, Dict, Optional
8
+
9
+ from ..base_metric import BaseMetric
10
+ from .types import StructuredInput, JSONInput, ValidationMode
11
+ from .validators import JSONValidator, YAMLValidator
12
+
13
+
14
+ class StructuredOutputScore(BaseMetric[StructuredInput]):
15
+ """
16
+ Comprehensive structured output evaluation.
17
+
18
+ Combines multiple aspects:
19
+ - Syntax validity (parseability)
20
+ - Schema compliance (matches expected structure)
21
+ - Field completeness (required fields present)
22
+ - Type correctness (values have correct types)
23
+ - Value accuracy (optional, if expected provided)
24
+
25
+ Score: 0.0 to 1.0 weighted combination of all aspects.
26
+
27
+ Example:
28
+ >>> metric = StructuredOutputScore()
29
+ >>> result = metric.evaluate([{
30
+ ... "response": '{"name": "Alice", "age": 25}',
31
+ ... "format": "json",
32
+ ... "schema": {
33
+ ... "type": "object",
34
+ ... "required": ["name", "age"],
35
+ ... "properties": {
36
+ ... "name": {"type": "string"},
37
+ ... "age": {"type": "integer"}
38
+ ... }
39
+ ... }
40
+ ... }])
41
+ """
42
+
43
+ @property
44
+ def metric_name(self) -> str:
45
+ return "structured_output_score"
46
+
47
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
48
+ super().__init__(config)
49
+ self.json_validator = JSONValidator()
50
+ self._yaml_validator = None
51
+
52
+ # Weights for different aspects
53
+ self.syntax_weight = self.config.get("syntax_weight", 0.2)
54
+ self.schema_weight = self.config.get("schema_weight", 0.3)
55
+ self.completeness_weight = self.config.get("completeness_weight", 0.25)
56
+ self.type_weight = self.config.get("type_weight", 0.15)
57
+ self.value_weight = self.config.get("value_weight", 0.1)
58
+
59
+ @property
60
+ def yaml_validator(self):
61
+ if self._yaml_validator is None:
62
+ try:
63
+ self._yaml_validator = YAMLValidator()
64
+ except ImportError:
65
+ return None
66
+ return self._yaml_validator
67
+
68
+ def _get_validator(self, format_name: str):
69
+ """Get validator for format."""
70
+ if format_name.lower() == "yaml":
71
+ if self.yaml_validator is None:
72
+ raise ImportError("PyYAML is required for YAML validation")
73
+ return self.yaml_validator
74
+ return self.json_validator
75
+
76
+ def compute_one(self, inputs: StructuredInput) -> Dict[str, Any]:
77
+ response = inputs.response
78
+ format_name = inputs.format
79
+ schema = inputs.schema
80
+ expected = inputs.expected
81
+ mode_str = inputs.mode
82
+
83
+ # Convert mode string to enum
84
+ try:
85
+ mode = ValidationMode(mode_str)
86
+ except ValueError:
87
+ mode = ValidationMode.COERCE
88
+
89
+ if not response or not response.strip():
90
+ return {
91
+ "output": 0.0,
92
+ "reason": "Empty response",
93
+ "breakdown": {
94
+ "syntax": 0.0,
95
+ "schema": 0.0,
96
+ "completeness": 0.0,
97
+ "types": 0.0,
98
+ "values": 0.0,
99
+ },
100
+ }
101
+
102
+ # Get validator
103
+ try:
104
+ validator = self._get_validator(format_name)
105
+ except ImportError as e:
106
+ return {
107
+ "output": 0.0,
108
+ "reason": str(e),
109
+ }
110
+
111
+ # Initialize scores
112
+ scores = {
113
+ "syntax": 0.0,
114
+ "schema": 0.0,
115
+ "completeness": 0.0,
116
+ "types": 0.0,
117
+ "values": 0.0,
118
+ }
119
+
120
+ # 1. Syntax validation
121
+ syntax_result = validator.validate_syntax(response)
122
+ if not syntax_result.syntax_valid:
123
+ return {
124
+ "output": 0.0,
125
+ "reason": f"Syntax error: {syntax_result.errors[0].message if syntax_result.errors else 'Unknown'}",
126
+ "breakdown": scores,
127
+ "errors": [e.dict() for e in syntax_result.errors],
128
+ }
129
+
130
+ scores["syntax"] = 1.0
131
+ parsed = syntax_result.parsed
132
+
133
+ # Track which dimensions are evaluable
134
+ evaluable = {"syntax": self.syntax_weight}
135
+
136
+ # 2. Schema validation (if schema provided)
137
+ if schema:
138
+ evaluable["schema"] = self.schema_weight
139
+ evaluable["completeness"] = self.completeness_weight
140
+ evaluable["types"] = self.type_weight
141
+
142
+ schema_result = validator.validate_schema(response, schema, mode)
143
+ scores["completeness"] = schema_result.completeness
144
+
145
+ if schema_result.schema_valid:
146
+ scores["schema"] = 1.0
147
+ scores["types"] = 1.0
148
+ else:
149
+ # Analyze errors for partial scores
150
+ type_errors = sum(1 for e in schema_result.errors if e.error_type == "type")
151
+ schema_errors = len(schema_result.errors) - type_errors
152
+
153
+ # Estimate total expected validations
154
+ total_fields = self._count_schema_fields(schema)
155
+
156
+ if total_fields > 0:
157
+ scores["schema"] = max(0.0, 1.0 - schema_errors / total_fields)
158
+ scores["types"] = max(0.0, 1.0 - type_errors / total_fields)
159
+
160
+ # 3. Value accuracy (if expected provided)
161
+ if expected is not None:
162
+ evaluable["values"] = self.value_weight
163
+
164
+ compare_result = validator.compare(response, expected, mode)
165
+ if compare_result.valid:
166
+ scores["values"] = 1.0
167
+ else:
168
+ # Calculate partial value match
169
+ value_errors = len(compare_result.errors)
170
+ total_values = self._count_values(expected)
171
+ scores["values"] = max(0.0, 1.0 - value_errors / max(total_values, 1))
172
+
173
+ # Calculate weighted overall score — only evaluable dimensions contribute
174
+ # Unevaluable dimensions score 0, not 1 (no data ≠ perfect)
175
+ overall = sum(evaluable[k] * scores[k] for k in evaluable)
176
+
177
+ return {
178
+ "output": round(overall, 4),
179
+ "reason": self._generate_reason(scores, evaluable),
180
+ "breakdown": {k: round(v, 4) for k, v in scores.items()},
181
+ "parsed": parsed,
182
+ }
183
+
184
+ def _count_schema_fields(self, schema: Dict[str, Any], depth: int = 0) -> int:
185
+ """Count fields in schema."""
186
+ if depth > 10:
187
+ return 1
188
+
189
+ count = 0
190
+ if schema.get("type") == "object":
191
+ properties = schema.get("properties", {})
192
+ count = len(properties)
193
+ for prop_schema in properties.values():
194
+ count += self._count_schema_fields(prop_schema, depth + 1)
195
+ elif schema.get("type") == "array":
196
+ items = schema.get("items", {})
197
+ count = 1 + self._count_schema_fields(items, depth + 1)
198
+
199
+ return max(count, 1)
200
+
201
+ def _count_values(self, data: Any) -> int:
202
+ """Count total values in data structure."""
203
+ if isinstance(data, dict):
204
+ return sum(self._count_values(v) for v in data.values())
205
+ elif isinstance(data, list):
206
+ return sum(self._count_values(item) for item in data)
207
+ else:
208
+ return 1
209
+
210
+ def _generate_reason(self, scores: Dict[str, float], evaluable: Dict[str, float]) -> str:
211
+ """Generate human-readable reason."""
212
+ issues = []
213
+ if scores["syntax"] < 1.0:
214
+ issues.append("syntax errors")
215
+ if "schema" in evaluable and scores["schema"] < 1.0:
216
+ issues.append("schema violations")
217
+ if "completeness" in evaluable and scores["completeness"] < 1.0:
218
+ issues.append("missing fields")
219
+ if "types" in evaluable and scores["types"] < 1.0:
220
+ issues.append("type errors")
221
+ if "values" in evaluable and scores["values"] < 1.0:
222
+ issues.append("value mismatches")
223
+
224
+ skipped = [k for k in ("schema", "completeness", "types", "values") if k not in evaluable]
225
+
226
+ if not issues and not skipped:
227
+ return "Fully valid structured output"
228
+ parts = []
229
+ if issues:
230
+ parts.append(f"Issues: {', '.join(issues)}")
231
+ if skipped:
232
+ parts.append(f"Not evaluated: {', '.join(skipped)}")
233
+ return "; ".join(parts) if parts else "Fully valid structured output"
234
+
235
+
236
+ class QuickStructuredCheck(BaseMetric[JSONInput]):
237
+ """
238
+ Fast, lightweight structured output check.
239
+
240
+ Quick validation that just checks:
241
+ 1. Is it valid JSON?
242
+ 2. Does it have the expected keys?
243
+ 3. Are the types roughly correct?
244
+
245
+ Score: 0.0 to 1.0
246
+ """
247
+
248
+ @property
249
+ def metric_name(self) -> str:
250
+ return "quick_structured_check"
251
+
252
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
253
+ super().__init__(config)
254
+ self.validator = JSONValidator()
255
+
256
+ def compute_one(self, inputs: JSONInput) -> Dict[str, Any]:
257
+ response = inputs.response
258
+ schema = inputs.schema
259
+ expected = inputs.expected
260
+
261
+ if not response or not response.strip():
262
+ return {"output": 0.0, "reason": "Empty response"}
263
+
264
+ # Quick syntax check
265
+ syntax_result = self.validator.validate_syntax(response)
266
+ if not syntax_result.syntax_valid:
267
+ return {"output": 0.0, "reason": "Invalid JSON"}
268
+
269
+ parsed = syntax_result.parsed
270
+ score = 0.5 # Base score for valid JSON
271
+
272
+ # Quick schema check
273
+ if schema:
274
+ required = schema.get("required", [])
275
+ if required and isinstance(parsed, dict):
276
+ present = sum(1 for f in required if f in parsed)
277
+ schema_score = present / len(required)
278
+ score = 0.5 + (0.5 * schema_score)
279
+ else:
280
+ score = 1.0
281
+ elif expected is not None:
282
+ # Quick key comparison
283
+ if isinstance(expected, dict) and isinstance(parsed, dict):
284
+ expected_keys = set(expected.keys())
285
+ actual_keys = set(parsed.keys())
286
+ if expected_keys:
287
+ overlap = len(expected_keys & actual_keys) / len(expected_keys)
288
+ score = 0.5 + (0.5 * overlap)
289
+ else:
290
+ score = 1.0
291
+ else:
292
+ score = 1.0 if type(expected) is type(parsed) else 0.5
293
+
294
+ return {
295
+ "output": round(score, 4),
296
+ "reason": "Valid JSON" + (" with expected structure" if score >= 0.8 else ""),
297
+ "parsed": parsed,
298
+ }
@@ -0,0 +1,108 @@
1
+ """
2
+ Input types for structured output validation metrics.
3
+ """
4
+
5
+ import warnings
6
+ from typing import Optional, Dict, Any, List
7
+ from pydantic import BaseModel, Field, ConfigDict
8
+ from enum import Enum
9
+
10
+ warnings.filterwarnings("ignore", message='Field name "schema" in .* shadows an attribute in parent "BaseModel"')
11
+
12
+
13
+ class ValidationMode(Enum):
14
+ """Validation strictness modes."""
15
+ STRICT = "strict" # Exact match, no extra fields, correct types
16
+ COERCE = "coerce" # Allow type coercion (str -> int, etc.)
17
+ LENIENT = "lenient" # Allow extra fields, flexible types
18
+
19
+
20
+ class JSONInput(BaseModel):
21
+ """Input for JSON validation metrics."""
22
+ model_config = ConfigDict(extra="allow")
23
+
24
+ response: str = Field(..., description="LLM-generated JSON string")
25
+ schema: Optional[Dict[str, Any]] = Field(
26
+ None, description="JSON Schema to validate against"
27
+ )
28
+ expected: Optional[Dict[str, Any]] = Field(
29
+ None, description="Expected JSON object for comparison"
30
+ )
31
+ mode: str = Field(
32
+ "coerce", description="Validation strictness: strict, coerce, lenient"
33
+ )
34
+
35
+
36
+ class PydanticInput(BaseModel):
37
+ """Input for Pydantic model validation."""
38
+ model_config = ConfigDict(extra="allow")
39
+
40
+ response: str = Field(..., description="LLM-generated JSON string")
41
+ model_class: Optional[str] = Field(
42
+ None, description="Fully qualified Pydantic model class name"
43
+ )
44
+ model_schema: Optional[Dict[str, Any]] = Field(
45
+ None, description="JSON Schema derived from Pydantic model"
46
+ )
47
+ mode: str = Field("coerce")
48
+
49
+
50
+ class YAMLInput(BaseModel):
51
+ """Input for YAML validation metrics."""
52
+ model_config = ConfigDict(extra="allow")
53
+
54
+ response: str = Field(..., description="LLM-generated YAML string")
55
+ schema: Optional[Dict[str, Any]] = Field(
56
+ None, description="JSON Schema to validate against"
57
+ )
58
+ expected: Optional[Dict[str, Any]] = Field(
59
+ None, description="Expected YAML content as dict"
60
+ )
61
+
62
+
63
+ class StructuredInput(BaseModel):
64
+ """Generic input for any structured format."""
65
+ model_config = ConfigDict(extra="allow")
66
+
67
+ response: str = Field(..., description="LLM-generated structured output")
68
+ format: str = Field("json", description="Format: json, xml, yaml, toml")
69
+ schema: Optional[Dict[str, Any]] = Field(None, description="Schema to validate")
70
+ expected: Optional[Any] = Field(None, description="Expected output")
71
+ mode: str = Field("coerce")
72
+
73
+
74
+ class ValidationError(BaseModel):
75
+ """Single validation error."""
76
+ path: str = Field(..., description="JSON path to error (e.g., '$.user.name')")
77
+ message: str = Field(..., description="Error description")
78
+ error_type: str = Field(..., description="Error type: syntax, type, missing, extra")
79
+ expected: Optional[Any] = Field(None, description="Expected value/type")
80
+ actual: Optional[Any] = Field(None, description="Actual value/type")
81
+
82
+ def dict(self, **kwargs):
83
+ """Convert to dictionary."""
84
+ return {
85
+ "path": self.path,
86
+ "message": self.message,
87
+ "error_type": self.error_type,
88
+ "expected": self.expected,
89
+ "actual": self.actual,
90
+ }
91
+
92
+
93
+ class ValidationResult(BaseModel):
94
+ """Complete validation result."""
95
+ model_config = ConfigDict(extra="allow")
96
+
97
+ valid: bool = Field(..., description="Overall validity")
98
+ errors: List[ValidationError] = Field(default_factory=list)
99
+ warnings: List[ValidationError] = Field(default_factory=list)
100
+
101
+ # Detailed scores
102
+ syntax_valid: bool = True
103
+ schema_valid: bool = True
104
+ type_valid: bool = True
105
+ completeness: float = 1.0 # 0-1, fraction of required fields present
106
+
107
+ # Parsed output (if successful)
108
+ parsed: Optional[Any] = None
@@ -0,0 +1,30 @@
1
+ """
2
+ Validators for structured output formats.
3
+ """
4
+
5
+ from .base import BaseValidator
6
+ from .json_validator import JSONValidator
7
+ from .pydantic_validator import PydanticValidator
8
+ from .yaml_validator import YAMLValidator
9
+
10
+ __all__ = [
11
+ "BaseValidator",
12
+ "JSONValidator",
13
+ "PydanticValidator",
14
+ "YAMLValidator",
15
+ ]
16
+
17
+ # Validator registry for easy lookup
18
+ VALIDATORS = {
19
+ "json": JSONValidator,
20
+ "yaml": YAMLValidator,
21
+ "pydantic": PydanticValidator,
22
+ }
23
+
24
+
25
+ def get_validator(format_name: str) -> BaseValidator:
26
+ """Get validator instance by format name."""
27
+ validator_class = VALIDATORS.get(format_name.lower())
28
+ if validator_class is None:
29
+ raise ValueError(f"Unknown format: {format_name}. Available: {list(VALIDATORS.keys())}")
30
+ return validator_class()
@@ -0,0 +1,189 @@
1
+ """
2
+ Base validator interface for structured output validation.
3
+ """
4
+
5
+ from abc import ABC, abstractmethod
6
+ from typing import Any, Dict, List
7
+ from ..types import ValidationResult, ValidationError, ValidationMode
8
+
9
+
10
+ class BaseValidator(ABC):
11
+ """Abstract base class for format validators."""
12
+
13
+ format_name: str = "unknown"
14
+
15
+ @abstractmethod
16
+ def validate_syntax(self, content: str) -> ValidationResult:
17
+ """
18
+ Validate syntax only (is it parseable?).
19
+
20
+ Returns:
21
+ ValidationResult with syntax_valid set
22
+ """
23
+ pass
24
+
25
+ @abstractmethod
26
+ def validate_schema(
27
+ self,
28
+ content: str,
29
+ schema: Dict[str, Any],
30
+ mode: ValidationMode = ValidationMode.COERCE,
31
+ ) -> ValidationResult:
32
+ """
33
+ Validate against a schema.
34
+
35
+ Args:
36
+ content: Raw string content
37
+ schema: Schema to validate against
38
+ mode: Validation strictness
39
+
40
+ Returns:
41
+ ValidationResult with full validation details
42
+ """
43
+ pass
44
+
45
+ @abstractmethod
46
+ def parse(self, content: str) -> Any:
47
+ """
48
+ Parse content into Python object.
49
+
50
+ Raises:
51
+ ValueError: If content cannot be parsed
52
+ """
53
+ pass
54
+
55
+ def compare(
56
+ self,
57
+ content: str,
58
+ expected: Any,
59
+ mode: ValidationMode = ValidationMode.COERCE,
60
+ ) -> ValidationResult:
61
+ """
62
+ Compare parsed content against expected value.
63
+
64
+ Default implementation - can be overridden.
65
+ """
66
+ try:
67
+ parsed = self.parse(content)
68
+ except Exception as e:
69
+ return ValidationResult(
70
+ valid=False,
71
+ syntax_valid=False,
72
+ errors=[ValidationError(
73
+ path="$",
74
+ message=str(e),
75
+ error_type="syntax",
76
+ )]
77
+ )
78
+
79
+ errors = self._compare_values(parsed, expected, "$", mode)
80
+
81
+ return ValidationResult(
82
+ valid=len(errors) == 0,
83
+ syntax_valid=True,
84
+ errors=errors,
85
+ parsed=parsed,
86
+ )
87
+
88
+ def _compare_values(
89
+ self,
90
+ actual: Any,
91
+ expected: Any,
92
+ path: str,
93
+ mode: ValidationMode,
94
+ ) -> List[ValidationError]:
95
+ """Recursively compare values."""
96
+ errors = []
97
+
98
+ # Handle None cases
99
+ if expected is None and actual is None:
100
+ return errors
101
+ if expected is None or actual is None:
102
+ if expected != actual:
103
+ errors.append(ValidationError(
104
+ path=path,
105
+ message="Value mismatch (None vs non-None)",
106
+ error_type="value",
107
+ expected=expected,
108
+ actual=actual,
109
+ ))
110
+ return errors
111
+
112
+ # Type comparison
113
+ if type(actual) is not type(expected):
114
+ if mode == ValidationMode.STRICT:
115
+ errors.append(ValidationError(
116
+ path=path,
117
+ message="Type mismatch",
118
+ error_type="type",
119
+ expected=type(expected).__name__,
120
+ actual=type(actual).__name__,
121
+ ))
122
+ return errors
123
+ elif mode == ValidationMode.COERCE:
124
+ # Try to coerce
125
+ try:
126
+ actual = type(expected)(actual)
127
+ except (ValueError, TypeError):
128
+ errors.append(ValidationError(
129
+ path=path,
130
+ message=f"Cannot coerce {type(actual).__name__} to {type(expected).__name__}",
131
+ error_type="type",
132
+ expected=type(expected).__name__,
133
+ actual=type(actual).__name__,
134
+ ))
135
+ return errors
136
+
137
+ # Dict comparison
138
+ if isinstance(expected, dict):
139
+ # Check for missing keys
140
+ for key in expected:
141
+ if key not in actual:
142
+ errors.append(ValidationError(
143
+ path=f"{path}.{key}",
144
+ message="Missing required field",
145
+ error_type="missing",
146
+ expected=key,
147
+ ))
148
+ else:
149
+ errors.extend(self._compare_values(
150
+ actual[key], expected[key], f"{path}.{key}", mode
151
+ ))
152
+
153
+ # Check for extra keys in strict mode
154
+ if mode == ValidationMode.STRICT:
155
+ for key in actual:
156
+ if key not in expected:
157
+ errors.append(ValidationError(
158
+ path=f"{path}.{key}",
159
+ message="Unexpected field",
160
+ error_type="extra",
161
+ actual=key,
162
+ ))
163
+
164
+ # List comparison
165
+ elif isinstance(expected, list):
166
+ if len(actual) != len(expected):
167
+ errors.append(ValidationError(
168
+ path=path,
169
+ message="Array length mismatch",
170
+ error_type="length",
171
+ expected=len(expected),
172
+ actual=len(actual),
173
+ ))
174
+ else:
175
+ for i, (a, e) in enumerate(zip(actual, expected)):
176
+ errors.extend(self._compare_values(a, e, f"{path}[{i}]", mode))
177
+
178
+ # Scalar comparison
179
+ else:
180
+ if actual != expected:
181
+ errors.append(ValidationError(
182
+ path=path,
183
+ message="Value mismatch",
184
+ error_type="value",
185
+ expected=expected,
186
+ actual=actual,
187
+ ))
188
+
189
+ return errors