agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,313 @@
1
+ """
2
+ Field Completeness Metric.
3
+
4
+ Measures the presence of required and optional fields in structured output.
5
+ """
6
+
7
+ from typing import Any, Dict, Optional, Set
8
+
9
+ from ..base_metric import BaseMetric
10
+ from .types import StructuredInput, JSONInput
11
+ from .validators import JSONValidator
12
+
13
+
14
+ class FieldCompleteness(BaseMetric[StructuredInput]):
15
+ """
16
+ Evaluates field completeness in structured output.
17
+
18
+ Measures:
19
+ - Required field presence (weighted heavily)
20
+ - Optional field presence (weighted lightly)
21
+ - Nested field coverage
22
+
23
+ Score: 0.0 (no required fields) to 1.0 (all fields present)
24
+
25
+ Example:
26
+ >>> metric = FieldCompleteness()
27
+ >>> result = metric.evaluate([{
28
+ ... "response": '{"name": "Alice", "email": "alice@example.com"}',
29
+ ... "format": "json",
30
+ ... "schema": {
31
+ ... "type": "object",
32
+ ... "required": ["name", "email", "age"],
33
+ ... "properties": {
34
+ ... "name": {"type": "string"},
35
+ ... "email": {"type": "string"},
36
+ ... "age": {"type": "integer"},
37
+ ... "phone": {"type": "string"}
38
+ ... }
39
+ ... }
40
+ ... }])
41
+ # Result: 0.67 (2/3 required fields present)
42
+ """
43
+
44
+ @property
45
+ def metric_name(self) -> str:
46
+ return "field_completeness"
47
+
48
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
49
+ super().__init__(config)
50
+ self.validator = JSONValidator()
51
+ self.required_weight = self.config.get("required_weight", 0.8)
52
+ self.optional_weight = self.config.get("optional_weight", 0.2)
53
+ self.include_nested = self.config.get("include_nested", True)
54
+
55
+ def compute_one(self, inputs: StructuredInput) -> Dict[str, Any]:
56
+ response = inputs.response
57
+ schema = inputs.schema
58
+
59
+ if not response or not response.strip():
60
+ return {
61
+ "output": 0.0,
62
+ "reason": "Empty response",
63
+ }
64
+
65
+ if not schema:
66
+ return {
67
+ "output": 0.0,
68
+ "reason": "No schema provided for field analysis",
69
+ }
70
+
71
+ # Parse response
72
+ syntax_result = self.validator.validate_syntax(response)
73
+ if not syntax_result.syntax_valid:
74
+ return {
75
+ "output": 0.0,
76
+ "reason": "Cannot parse response",
77
+ "errors": [e.dict() for e in syntax_result.errors],
78
+ }
79
+
80
+ parsed = syntax_result.parsed
81
+
82
+ # Analyze field presence
83
+ analysis = self._analyze_fields(parsed, schema, "$")
84
+
85
+ # Calculate score
86
+ required_score = analysis["required_present"] / max(analysis["required_total"], 1)
87
+ optional_score = analysis["optional_present"] / max(analysis["optional_total"], 1)
88
+
89
+ # Weight the scores — only include components that exist
90
+ has_required = analysis["required_total"] > 0
91
+ has_optional = analysis["optional_total"] > 0
92
+
93
+ if has_required and has_optional:
94
+ score = (
95
+ self.required_weight * required_score +
96
+ self.optional_weight * optional_score
97
+ )
98
+ elif has_required:
99
+ score = required_score
100
+ elif has_optional:
101
+ score = optional_score
102
+ else:
103
+ score = 1.0
104
+
105
+ return {
106
+ "output": round(score, 4),
107
+ "reason": f"{analysis['required_present']}/{analysis['required_total']} required, {analysis['optional_present']}/{analysis['optional_total']} optional fields",
108
+ "required_fields": {
109
+ "present": analysis["required_present"],
110
+ "total": analysis["required_total"],
111
+ "missing": analysis["missing_required"],
112
+ },
113
+ "optional_fields": {
114
+ "present": analysis["optional_present"],
115
+ "total": analysis["optional_total"],
116
+ "missing": analysis["missing_optional"],
117
+ },
118
+ "completeness": required_score,
119
+ "parsed": parsed,
120
+ }
121
+
122
+ def _analyze_fields(
123
+ self,
124
+ data: Any,
125
+ schema: Dict[str, Any],
126
+ path: str,
127
+ ) -> Dict[str, Any]:
128
+ """Analyze field presence recursively."""
129
+ result = {
130
+ "required_present": 0,
131
+ "required_total": 0,
132
+ "optional_present": 0,
133
+ "optional_total": 0,
134
+ "missing_required": [],
135
+ "missing_optional": [],
136
+ }
137
+
138
+ if schema.get("type") != "object" or not isinstance(data, dict):
139
+ return result
140
+
141
+ properties = schema.get("properties", {})
142
+ required = set(schema.get("required", []))
143
+
144
+ for field, field_schema in properties.items():
145
+ field_path = f"{path}.{field}"
146
+ is_required = field in required
147
+ is_present = field in data
148
+
149
+ if is_required:
150
+ result["required_total"] += 1
151
+ if is_present:
152
+ result["required_present"] += 1
153
+ else:
154
+ result["missing_required"].append(field_path)
155
+ else:
156
+ result["optional_total"] += 1
157
+ if is_present:
158
+ result["optional_present"] += 1
159
+ else:
160
+ result["missing_optional"].append(field_path)
161
+
162
+ # Recursively analyze nested objects
163
+ if self.include_nested and is_present and field_schema.get("type") == "object":
164
+ nested_result = self._analyze_fields(
165
+ data[field], field_schema, field_path
166
+ )
167
+ result["required_present"] += nested_result["required_present"]
168
+ result["required_total"] += nested_result["required_total"]
169
+ result["optional_present"] += nested_result["optional_present"]
170
+ result["optional_total"] += nested_result["optional_total"]
171
+ result["missing_required"].extend(nested_result["missing_required"])
172
+ result["missing_optional"].extend(nested_result["missing_optional"])
173
+
174
+ return result
175
+
176
+
177
+ class RequiredFieldsOnly(BaseMetric[StructuredInput]):
178
+ """
179
+ Simple metric that only checks required field presence.
180
+
181
+ Returns the fraction of required fields present.
182
+
183
+ Score: 0.0 to 1.0
184
+ """
185
+
186
+ @property
187
+ def metric_name(self) -> str:
188
+ return "required_fields"
189
+
190
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
191
+ super().__init__(config)
192
+ self.validator = JSONValidator()
193
+
194
+ def compute_one(self, inputs: StructuredInput) -> Dict[str, Any]:
195
+ response = inputs.response
196
+ schema = inputs.schema
197
+
198
+ if not response or not response.strip():
199
+ return {"output": 0.0, "reason": "Empty response"}
200
+
201
+ if not schema:
202
+ return {"output": 0.0, "reason": "No schema provided"}
203
+
204
+ # Parse response
205
+ syntax_result = self.validator.validate_syntax(response)
206
+ if not syntax_result.syntax_valid:
207
+ return {"output": 0.0, "reason": "Cannot parse response"}
208
+
209
+ parsed = syntax_result.parsed
210
+
211
+ # Get required fields
212
+ required = schema.get("required", [])
213
+ if not required:
214
+ return {
215
+ "output": 1.0,
216
+ "reason": "No required fields in schema",
217
+ "parsed": parsed,
218
+ }
219
+
220
+ # Check presence
221
+ present = [f for f in required if isinstance(parsed, dict) and f in parsed]
222
+ missing = [f for f in required if f not in present]
223
+
224
+ score = len(present) / len(required)
225
+
226
+ return {
227
+ "output": round(score, 4),
228
+ "reason": f"{len(present)}/{len(required)} required fields present",
229
+ "present_fields": present,
230
+ "missing_fields": missing,
231
+ "parsed": parsed,
232
+ }
233
+
234
+
235
+ class FieldCoverage(BaseMetric[JSONInput]):
236
+ """
237
+ Measures field coverage comparing response to expected output.
238
+
239
+ Compares actual fields present vs expected fields without validating values.
240
+
241
+ Score: 0.0 to 1.0 (fraction of expected fields present)
242
+ """
243
+
244
+ @property
245
+ def metric_name(self) -> str:
246
+ return "field_coverage"
247
+
248
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
249
+ super().__init__(config)
250
+ self.validator = JSONValidator()
251
+ self.include_nested = self.config.get("include_nested", True)
252
+
253
+ def compute_one(self, inputs: JSONInput) -> Dict[str, Any]:
254
+ response = inputs.response
255
+ expected = inputs.expected
256
+
257
+ if not response or not response.strip():
258
+ return {"output": 0.0, "reason": "Empty response"}
259
+
260
+ if expected is None:
261
+ return {"output": 0.0, "reason": "No expected output provided"}
262
+
263
+ # Parse response
264
+ syntax_result = self.validator.validate_syntax(response)
265
+ if not syntax_result.syntax_valid:
266
+ return {"output": 0.0, "reason": "Cannot parse response"}
267
+
268
+ parsed = syntax_result.parsed
269
+
270
+ # Extract fields from both
271
+ expected_fields = self._extract_fields(expected, "$")
272
+ actual_fields = self._extract_fields(parsed, "$")
273
+
274
+ # Calculate coverage
275
+ if not expected_fields:
276
+ return {
277
+ "output": 1.0,
278
+ "reason": "No fields expected",
279
+ "parsed": parsed,
280
+ }
281
+
282
+ covered = expected_fields & actual_fields
283
+ missing = expected_fields - actual_fields
284
+ extra = actual_fields - expected_fields
285
+
286
+ score = len(covered) / len(expected_fields)
287
+
288
+ return {
289
+ "output": round(score, 4),
290
+ "reason": f"{len(covered)}/{len(expected_fields)} expected fields present",
291
+ "covered_fields": list(covered),
292
+ "missing_fields": list(missing),
293
+ "extra_fields": list(extra),
294
+ "parsed": parsed,
295
+ }
296
+
297
+ def _extract_fields(self, data: Any, path: str) -> Set[str]:
298
+ """Extract all field paths from data."""
299
+ fields = set()
300
+
301
+ if isinstance(data, dict):
302
+ for key, value in data.items():
303
+ field_path = f"{path}.{key}"
304
+ fields.add(field_path)
305
+
306
+ if self.include_nested:
307
+ fields.update(self._extract_fields(value, field_path))
308
+
309
+ elif isinstance(data, list) and self.include_nested:
310
+ for i, item in enumerate(data):
311
+ fields.update(self._extract_fields(item, f"{path}[{i}]"))
312
+
313
+ return fields
@@ -0,0 +1,366 @@
1
+ """
2
+ Hierarchy Score Metric.
3
+
4
+ Tree-based structural comparison using edit distance concepts.
5
+ Inspired by STED (Structural Tree Edit Distance) for comparing hierarchical structures.
6
+ """
7
+
8
+ from typing import Any, Dict, List, Optional
9
+
10
+ from ..base_metric import BaseMetric
11
+ from .types import JSONInput
12
+ from .validators import JSONValidator
13
+
14
+
15
+ class HierarchyScore(BaseMetric[JSONInput]):
16
+ """
17
+ Evaluates structural similarity using tree-based comparison.
18
+
19
+ Compares the structural hierarchy of response vs expected:
20
+ - Tree structure matching
21
+ - Key path similarity
22
+ - Depth alignment
23
+ - Array structure matching
24
+
25
+ Inspired by STED (Structural Tree Edit Distance) concepts for
26
+ comparing hierarchical JSON structures.
27
+
28
+ Score: 0.0 (completely different structure) to 1.0 (identical structure)
29
+
30
+ Example:
31
+ >>> metric = HierarchyScore()
32
+ >>> result = metric.evaluate([{
33
+ ... "response": '{"user": {"name": "Alice"}, "items": [1, 2]}',
34
+ ... "expected": {"user": {"name": "Bob", "email": "b@ex.com"}, "items": [1, 2, 3]}
35
+ ... }])
36
+ # Compares structure, not values
37
+ """
38
+
39
+ @property
40
+ def metric_name(self) -> str:
41
+ return "hierarchy_score"
42
+
43
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
44
+ super().__init__(config)
45
+ self.validator = JSONValidator()
46
+ self.key_weight = self.config.get("key_weight", 0.5)
47
+ self.type_weight = self.config.get("type_weight", 0.3)
48
+ self.depth_weight = self.config.get("depth_weight", 0.2)
49
+ self.max_depth = self.config.get("max_depth", 10)
50
+
51
+ def compute_one(self, inputs: JSONInput) -> Dict[str, Any]:
52
+ response = inputs.response
53
+ expected = inputs.expected
54
+
55
+ if not response or not response.strip():
56
+ return {"output": 0.0, "reason": "Empty response"}
57
+
58
+ if expected is None:
59
+ return {"output": 0.0, "reason": "No expected structure provided"}
60
+
61
+ # Parse response
62
+ syntax_result = self.validator.validate_syntax(response)
63
+ if not syntax_result.syntax_valid:
64
+ return {"output": 0.0, "reason": "Cannot parse response"}
65
+
66
+ parsed = syntax_result.parsed
67
+
68
+ # Build structure fingerprints
69
+ expected_structure = self._build_structure(expected, "", 0)
70
+ actual_structure = self._build_structure(parsed, "", 0)
71
+
72
+ # Compare structures
73
+ similarity = self._compare_structures(expected_structure, actual_structure)
74
+
75
+ return {
76
+ "output": round(similarity["overall"], 4),
77
+ "reason": f"Structure similarity: keys={similarity['key_similarity']:.2f}, types={similarity['type_similarity']:.2f}, depth={similarity['depth_similarity']:.2f}",
78
+ "key_similarity": round(similarity["key_similarity"], 4),
79
+ "type_similarity": round(similarity["type_similarity"], 4),
80
+ "depth_similarity": round(similarity["depth_similarity"], 4),
81
+ "expected_depth": expected_structure["max_depth"],
82
+ "actual_depth": actual_structure["max_depth"],
83
+ "missing_keys": list(similarity["missing_keys"]),
84
+ "extra_keys": list(similarity["extra_keys"]),
85
+ "parsed": parsed,
86
+ }
87
+
88
+ def _build_structure(
89
+ self,
90
+ data: Any,
91
+ path: str,
92
+ depth: int,
93
+ ) -> Dict[str, Any]:
94
+ """Build structure fingerprint for comparison."""
95
+ structure = {
96
+ "keys": set(),
97
+ "types": {},
98
+ "depths": {},
99
+ "max_depth": depth,
100
+ "array_shapes": {},
101
+ }
102
+
103
+ if depth > self.max_depth:
104
+ return structure
105
+
106
+ if isinstance(data, dict):
107
+ for key, value in data.items():
108
+ key_path = f"{path}.{key}" if path else key
109
+ structure["keys"].add(key_path)
110
+ structure["types"][key_path] = self._get_type_name(value)
111
+ structure["depths"][key_path] = depth
112
+
113
+ # Recurse into nested structures
114
+ nested = self._build_structure(value, key_path, depth + 1)
115
+ structure["keys"].update(nested["keys"])
116
+ structure["types"].update(nested["types"])
117
+ structure["depths"].update(nested["depths"])
118
+ structure["max_depth"] = max(structure["max_depth"], nested["max_depth"])
119
+ structure["array_shapes"].update(nested["array_shapes"])
120
+
121
+ elif isinstance(data, list):
122
+ structure["array_shapes"][path] = len(data)
123
+ if data:
124
+ # Sample first element for structure
125
+ sample_path = f"{path}[]"
126
+ structure["keys"].add(sample_path)
127
+ structure["types"][sample_path] = self._get_type_name(data[0])
128
+ structure["depths"][sample_path] = depth
129
+
130
+ if isinstance(data[0], (dict, list)):
131
+ nested = self._build_structure(data[0], sample_path, depth + 1)
132
+ structure["keys"].update(nested["keys"])
133
+ structure["types"].update(nested["types"])
134
+ structure["depths"].update(nested["depths"])
135
+ structure["max_depth"] = max(structure["max_depth"], nested["max_depth"])
136
+ structure["array_shapes"].update(nested["array_shapes"])
137
+
138
+ return structure
139
+
140
+ def _get_type_name(self, value: Any) -> str:
141
+ """Get JSON-like type name."""
142
+ if value is None:
143
+ return "null"
144
+ elif isinstance(value, bool):
145
+ return "boolean"
146
+ elif isinstance(value, int):
147
+ return "integer"
148
+ elif isinstance(value, float):
149
+ return "number"
150
+ elif isinstance(value, str):
151
+ return "string"
152
+ elif isinstance(value, list):
153
+ return "array"
154
+ elif isinstance(value, dict):
155
+ return "object"
156
+ else:
157
+ return "unknown"
158
+
159
+ def _compare_structures(
160
+ self,
161
+ expected: Dict[str, Any],
162
+ actual: Dict[str, Any],
163
+ ) -> Dict[str, Any]:
164
+ """Compare two structure fingerprints."""
165
+ expected_keys = expected["keys"]
166
+ actual_keys = actual["keys"]
167
+
168
+ # Key similarity (Jaccard)
169
+ if expected_keys or actual_keys:
170
+ intersection = expected_keys & actual_keys
171
+ union = expected_keys | actual_keys
172
+ key_similarity = len(intersection) / len(union) if union else 1.0
173
+ else:
174
+ key_similarity = 1.0
175
+
176
+ # Type similarity for matching keys
177
+ matching_keys = expected_keys & actual_keys
178
+ if matching_keys:
179
+ type_matches = sum(
180
+ 1 for k in matching_keys
181
+ if expected["types"].get(k) == actual["types"].get(k)
182
+ )
183
+ type_similarity = type_matches / len(matching_keys)
184
+ else:
185
+ type_similarity = 0.0 if expected_keys else 1.0
186
+
187
+ # Depth similarity
188
+ max_expected = expected["max_depth"]
189
+ max_actual = actual["max_depth"]
190
+ if max_expected == 0 and max_actual == 0:
191
+ depth_similarity = 1.0
192
+ else:
193
+ depth_similarity = 1.0 - abs(max_expected - max_actual) / max(max_expected, max_actual, 1)
194
+
195
+ # Overall weighted score
196
+ overall = (
197
+ self.key_weight * key_similarity +
198
+ self.type_weight * type_similarity +
199
+ self.depth_weight * depth_similarity
200
+ )
201
+
202
+ return {
203
+ "overall": overall,
204
+ "key_similarity": key_similarity,
205
+ "type_similarity": type_similarity,
206
+ "depth_similarity": depth_similarity,
207
+ "missing_keys": expected_keys - actual_keys,
208
+ "extra_keys": actual_keys - expected_keys,
209
+ }
210
+
211
+
212
+ class TreeEditDistance(BaseMetric[JSONInput]):
213
+ """
214
+ Computes normalized tree edit distance between structures.
215
+
216
+ Based on simplified tree edit distance where operations are:
217
+ - Insert node
218
+ - Delete node
219
+ - Rename node (change key or value)
220
+
221
+ Score: 0.0 (identical) to 1.0 (completely different)
222
+ Note: This returns DISTANCE, so lower is better.
223
+ """
224
+
225
+ @property
226
+ def metric_name(self) -> str:
227
+ return "tree_edit_distance"
228
+
229
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
230
+ super().__init__(config)
231
+ self.validator = JSONValidator()
232
+ self.insert_cost = self.config.get("insert_cost", 1.0)
233
+ self.delete_cost = self.config.get("delete_cost", 1.0)
234
+ self.rename_cost = self.config.get("rename_cost", 0.5)
235
+
236
+ def compute_one(self, inputs: JSONInput) -> Dict[str, Any]:
237
+ response = inputs.response
238
+ expected = inputs.expected
239
+
240
+ if not response or not response.strip():
241
+ return {"output": 1.0, "reason": "Empty response (max distance)"}
242
+
243
+ if expected is None:
244
+ return {"output": 1.0, "reason": "No expected structure provided"}
245
+
246
+ # Parse response
247
+ syntax_result = self.validator.validate_syntax(response)
248
+ if not syntax_result.syntax_valid:
249
+ return {"output": 1.0, "reason": "Cannot parse response"}
250
+
251
+ parsed = syntax_result.parsed
252
+
253
+ # Compute tree operations needed
254
+ operations = self._compute_operations(expected, parsed, "$")
255
+
256
+ # Calculate total cost
257
+ total_cost = sum(op["cost"] for op in operations)
258
+
259
+ # Normalize by tree size
260
+ expected_size = self._count_nodes(expected)
261
+ actual_size = self._count_nodes(parsed)
262
+ max_size = max(expected_size, actual_size, 1)
263
+
264
+ normalized_distance = min(total_cost / max_size, 1.0)
265
+
266
+ return {
267
+ "output": round(normalized_distance, 4),
268
+ "reason": f"{len(operations)} operations (cost={total_cost:.2f})",
269
+ "operations": operations[:10], # Limit for readability
270
+ "total_operations": len(operations),
271
+ "total_cost": round(total_cost, 4),
272
+ "expected_nodes": expected_size,
273
+ "actual_nodes": actual_size,
274
+ "parsed": parsed,
275
+ }
276
+
277
+ def _compute_operations(
278
+ self,
279
+ expected: Any,
280
+ actual: Any,
281
+ path: str,
282
+ ) -> List[Dict[str, Any]]:
283
+ """Compute edit operations needed to transform actual to expected."""
284
+ operations = []
285
+
286
+ # Handle type mismatches
287
+ if type(expected) is not type(actual):
288
+ operations.append({
289
+ "type": "replace",
290
+ "path": path,
291
+ "from_type": type(actual).__name__,
292
+ "to_type": type(expected).__name__,
293
+ "cost": self.rename_cost,
294
+ })
295
+ return operations
296
+
297
+ if isinstance(expected, dict):
298
+ expected_keys = set(expected.keys())
299
+ actual_keys = set(actual.keys())
300
+
301
+ # Missing keys (need insert)
302
+ for key in expected_keys - actual_keys:
303
+ operations.append({
304
+ "type": "insert",
305
+ "path": f"{path}.{key}",
306
+ "cost": self.insert_cost,
307
+ })
308
+
309
+ # Extra keys (need delete)
310
+ for key in actual_keys - expected_keys:
311
+ operations.append({
312
+ "type": "delete",
313
+ "path": f"{path}.{key}",
314
+ "cost": self.delete_cost,
315
+ })
316
+
317
+ # Recurse into common keys
318
+ for key in expected_keys & actual_keys:
319
+ operations.extend(
320
+ self._compute_operations(expected[key], actual[key], f"{path}.{key}")
321
+ )
322
+
323
+ elif isinstance(expected, list):
324
+ # Simple length-based comparison for arrays
325
+ len_diff = abs(len(expected) - len(actual))
326
+ for i in range(len_diff):
327
+ if len(expected) > len(actual):
328
+ operations.append({
329
+ "type": "insert",
330
+ "path": f"{path}[{len(actual) + i}]",
331
+ "cost": self.insert_cost,
332
+ })
333
+ else:
334
+ operations.append({
335
+ "type": "delete",
336
+ "path": f"{path}[{len(expected) + i}]",
337
+ "cost": self.delete_cost,
338
+ })
339
+
340
+ # Compare common elements
341
+ for i, (exp_item, act_item) in enumerate(zip(expected, actual)):
342
+ operations.extend(
343
+ self._compute_operations(exp_item, act_item, f"{path}[{i}]")
344
+ )
345
+
346
+ else:
347
+ # Scalar comparison
348
+ if expected != actual:
349
+ operations.append({
350
+ "type": "rename",
351
+ "path": path,
352
+ "from": str(actual)[:50],
353
+ "to": str(expected)[:50],
354
+ "cost": self.rename_cost,
355
+ })
356
+
357
+ return operations
358
+
359
+ def _count_nodes(self, data: Any) -> int:
360
+ """Count total nodes in tree."""
361
+ if isinstance(data, dict):
362
+ return 1 + sum(self._count_nodes(v) for v in data.values())
363
+ elif isinstance(data, list):
364
+ return 1 + sum(self._count_nodes(item) for item in data)
365
+ else:
366
+ return 1