agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,573 @@
1
+ """
2
+ Function Calling Evaluation Metrics.
3
+
4
+ AST-based, deterministic evaluation of LLM function calling.
5
+ Provides sub-10ms evaluation latency without LLM-as-judge dependency.
6
+ """
7
+
8
+ import ast
9
+ import json
10
+ from typing import Any, Dict, List, Optional, Union
11
+
12
+ from ..base_metric import BaseMetric
13
+ from .types import FunctionCallInput, FunctionCall
14
+
15
+
16
+ def _parse_function_call(call: Union[FunctionCall, Dict, str, None]) -> Optional[FunctionCall]:
17
+ """Parse various input formats into a FunctionCall object."""
18
+ if call is None:
19
+ return None
20
+ if isinstance(call, FunctionCall):
21
+ return call
22
+ if isinstance(call, dict):
23
+ # Get arguments, handling both dict and JSON string formats (OpenAI-style)
24
+ arguments = call.get("arguments", call.get("parameters", call.get("input", {})))
25
+ if isinstance(arguments, str):
26
+ try:
27
+ arguments = json.loads(arguments)
28
+ except json.JSONDecodeError:
29
+ arguments = {}
30
+ return FunctionCall(
31
+ name=call.get("name", call.get("function", "")),
32
+ arguments=arguments if isinstance(arguments, dict) else {}
33
+ )
34
+ if isinstance(call, str):
35
+ try:
36
+ parsed = json.loads(call)
37
+ return _parse_function_call(parsed)
38
+ except json.JSONDecodeError:
39
+ # Try to parse as Python AST (e.g., "get_weather(city='NYC')")
40
+ return _parse_ast_call(call)
41
+ return None
42
+
43
+
44
+ def _parse_ast_call(call_str: str) -> Optional[FunctionCall]:
45
+ """Parse a function call string using AST."""
46
+ try:
47
+ # Wrap in expression for parsing
48
+ tree = ast.parse(call_str.strip(), mode='eval')
49
+ if isinstance(tree.body, ast.Call):
50
+ call_node = tree.body
51
+ func_name = ""
52
+ if isinstance(call_node.func, ast.Name):
53
+ func_name = call_node.func.id
54
+ elif isinstance(call_node.func, ast.Attribute):
55
+ func_name = call_node.func.attr
56
+
57
+ # Extract arguments
58
+ arguments = {}
59
+
60
+ # Handle positional arguments
61
+ for i, arg in enumerate(call_node.args):
62
+ arguments[f"__positional_{i}"] = _ast_to_value(arg)
63
+
64
+ # Handle keyword arguments
65
+ for kw in call_node.keywords:
66
+ if kw.arg:
67
+ arguments[kw.arg] = _ast_to_value(kw.value)
68
+
69
+ return FunctionCall(name=func_name, arguments=arguments)
70
+ except (SyntaxError, ValueError):
71
+ pass
72
+ return None
73
+
74
+
75
+ def _ast_to_value(node: ast.expr) -> Any:
76
+ """Convert an AST node to a Python value."""
77
+ # ast.Constant handles strings, numbers, booleans, None in Python 3.8+
78
+ if isinstance(node, ast.Constant):
79
+ return node.value
80
+ if isinstance(node, ast.List):
81
+ return [_ast_to_value(elt) for elt in node.elts]
82
+ if isinstance(node, ast.Dict):
83
+ return {
84
+ _ast_to_value(k): _ast_to_value(v)
85
+ for k, v in zip(node.keys, node.values)
86
+ if k is not None
87
+ }
88
+ if isinstance(node, ast.Name):
89
+ # Handle True, False, None as names (fallback)
90
+ if node.id == "True":
91
+ return True
92
+ elif node.id == "False":
93
+ return False
94
+ elif node.id == "None":
95
+ return None
96
+ return node.id
97
+ if isinstance(node, ast.Tuple):
98
+ return tuple(_ast_to_value(elt) for elt in node.elts)
99
+ if isinstance(node, ast.Set):
100
+ return {_ast_to_value(elt) for elt in node.elts}
101
+ return str(node)
102
+
103
+
104
+ def _parse_function_calls(
105
+ calls: Union[FunctionCall, List[FunctionCall], Dict, str, List[Dict], None]
106
+ ) -> List[FunctionCall]:
107
+ """Parse various input formats into a list of FunctionCall objects."""
108
+ if calls is None:
109
+ return []
110
+ if isinstance(calls, list):
111
+ return [_parse_function_call(c) for c in calls if _parse_function_call(c)]
112
+ parsed = _parse_function_call(calls)
113
+ return [parsed] if parsed else []
114
+
115
+
116
+ def _types_compatible(actual: Any, expected: Any, strict: bool = False) -> bool:
117
+ """Check if types are compatible."""
118
+ if strict:
119
+ return type(actual) is type(expected)
120
+
121
+ # Flexible type checking
122
+ if actual is None or expected is None:
123
+ return actual == expected
124
+
125
+ # Bool guard: bool is a subclass of int in Python, so must check before numeric
126
+ if isinstance(actual, bool) or isinstance(expected, bool):
127
+ return isinstance(actual, bool) and isinstance(expected, bool)
128
+
129
+ # Numeric compatibility (int/float)
130
+ if isinstance(actual, (int, float)) and isinstance(expected, (int, float)):
131
+ return True
132
+
133
+ # String compatibility
134
+ if isinstance(actual, str) and isinstance(expected, str):
135
+ return True
136
+
137
+ # List compatibility
138
+ if isinstance(actual, list) and isinstance(expected, list):
139
+ return True
140
+
141
+ # Dict compatibility
142
+ if isinstance(actual, dict) and isinstance(expected, dict):
143
+ return True
144
+
145
+ return type(actual) is type(expected)
146
+
147
+
148
+ def _values_equal(actual: Any, expected: Any, strict_type: bool = False) -> bool:
149
+ """Check if values are equal with optional type flexibility."""
150
+ if not _types_compatible(actual, expected, strict_type):
151
+ return False
152
+
153
+ # For numeric types, allow float/int comparison
154
+ if isinstance(actual, (int, float)) and isinstance(expected, (int, float)):
155
+ return float(actual) == float(expected)
156
+
157
+ # For strings, exact match (case-sensitive)
158
+ if isinstance(actual, str) and isinstance(expected, str):
159
+ return actual == expected
160
+
161
+ # For lists, recursive comparison
162
+ if isinstance(actual, list) and isinstance(expected, list):
163
+ if len(actual) != len(expected):
164
+ return False
165
+ return all(_values_equal(a, e, strict_type) for a, e in zip(actual, expected))
166
+
167
+ # For dicts, recursive comparison
168
+ if isinstance(actual, dict) and isinstance(expected, dict):
169
+ if set(actual.keys()) != set(expected.keys()):
170
+ return False
171
+ return all(
172
+ _values_equal(actual[k], expected[k], strict_type)
173
+ for k in expected.keys()
174
+ )
175
+
176
+ return actual == expected
177
+
178
+
179
+ class FunctionNameMatch(BaseMetric[FunctionCallInput]):
180
+ """
181
+ Evaluates if the function name matches the expected name.
182
+
183
+ Returns 1.0 if names match, 0.0 otherwise.
184
+ Fast, deterministic metric (~1ms).
185
+ """
186
+
187
+ @property
188
+ def metric_name(self) -> str:
189
+ return "function_name_match"
190
+
191
+ def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
192
+ actual = _parse_function_call(inputs.response)
193
+ expected = _parse_function_call(inputs.expected_response)
194
+
195
+ if actual is None:
196
+ return {
197
+ "output": 0.0,
198
+ "reason": "Could not parse actual function call from response."
199
+ }
200
+
201
+ if expected is None:
202
+ return {
203
+ "output": 0.0,
204
+ "reason": "Could not parse expected function call. 'expected_response' is required."
205
+ }
206
+
207
+ if actual.name == expected.name:
208
+ return {
209
+ "output": 1.0,
210
+ "reason": f"Function name '{actual.name}' matches expected."
211
+ }
212
+
213
+ return {
214
+ "output": 0.0,
215
+ "reason": f"Function name mismatch: got '{actual.name}', expected '{expected.name}'."
216
+ }
217
+
218
+
219
+ class ParameterValidation(BaseMetric[FunctionCallInput]):
220
+ """
221
+ Validates function call parameters against a schema.
222
+
223
+ Checks:
224
+ - Required parameters are present
225
+ - Parameter types match specification
226
+ - Enum constraints are satisfied
227
+
228
+ Returns score from 0.0 to 1.0 based on validation success.
229
+ """
230
+
231
+ @property
232
+ def metric_name(self) -> str:
233
+ return "parameter_validation"
234
+
235
+ def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
236
+ actual = _parse_function_call(inputs.response)
237
+
238
+ if actual is None:
239
+ return {
240
+ "output": 0.0,
241
+ "reason": "Could not parse function call from response."
242
+ }
243
+
244
+ if not inputs.function_definitions:
245
+ return {
246
+ "output": 0.0,
247
+ "reason": "No function definitions provided for validation."
248
+ }
249
+
250
+ # Find the matching function definition
251
+ func_def = None
252
+ for fd in inputs.function_definitions:
253
+ if fd.name == actual.name:
254
+ func_def = fd
255
+ break
256
+
257
+ if func_def is None:
258
+ return {
259
+ "output": 0.0,
260
+ "reason": f"Function '{actual.name}' not found in definitions."
261
+ }
262
+
263
+ errors = []
264
+ total_checks = 0
265
+ passed_checks = 0
266
+
267
+ for param_spec in func_def.parameters:
268
+ total_checks += 1
269
+
270
+ # Check required parameters
271
+ if param_spec.required and param_spec.name not in actual.arguments:
272
+ errors.append(f"Missing required parameter: {param_spec.name}")
273
+ continue
274
+
275
+ if param_spec.name in actual.arguments:
276
+ value = actual.arguments[param_spec.name]
277
+
278
+ # Type checking
279
+ if not self._check_type(value, param_spec.type, inputs.strict_type_check):
280
+ errors.append(
281
+ f"Parameter '{param_spec.name}' has wrong type: "
282
+ f"expected {param_spec.type}, got {type(value).__name__}"
283
+ )
284
+ continue
285
+
286
+ # Enum checking
287
+ if param_spec.enum and value not in param_spec.enum:
288
+ errors.append(
289
+ f"Parameter '{param_spec.name}' value '{value}' not in allowed values: {param_spec.enum}"
290
+ )
291
+ continue
292
+
293
+ passed_checks += 1
294
+ else:
295
+ # Optional parameter not provided - that's fine
296
+ passed_checks += 1
297
+
298
+ # Check for extra parameters
299
+ if not inputs.ignore_extra_params:
300
+ expected_params = {p.name for p in func_def.parameters}
301
+ extra_params = set(actual.arguments.keys()) - expected_params
302
+ if extra_params:
303
+ total_checks += len(extra_params)
304
+ errors.append(f"Unexpected parameters: {', '.join(extra_params)}")
305
+
306
+ if total_checks == 0:
307
+ return {"output": 1.0, "reason": "No parameters to validate."}
308
+
309
+ score = passed_checks / total_checks if total_checks > 0 else 1.0
310
+
311
+ if errors:
312
+ return {
313
+ "output": round(score, 4),
314
+ "reason": "; ".join(errors)
315
+ }
316
+
317
+ return {
318
+ "output": 1.0,
319
+ "reason": f"All {total_checks} parameter checks passed."
320
+ }
321
+
322
+ def _check_type(self, value: Any, expected_type: str, strict: bool) -> bool:
323
+ """Check if a value matches the expected type."""
324
+ type_map = {
325
+ "string": (str,),
326
+ "integer": (int,) if strict else (int, float),
327
+ "number": (int, float),
328
+ "boolean": (bool,),
329
+ "array": (list,),
330
+ "object": (dict,),
331
+ "null": (type(None),),
332
+ }
333
+
334
+ expected_types = type_map.get(expected_type.lower(), (str,))
335
+
336
+ # Special case: strict integer check
337
+ if expected_type.lower() == "integer" and strict:
338
+ return isinstance(value, int) and not isinstance(value, bool)
339
+
340
+ # Boolean should not match int in Python
341
+ if isinstance(value, bool) and expected_type.lower() != "boolean":
342
+ return False
343
+
344
+ return isinstance(value, expected_types)
345
+
346
+
347
+ class FunctionCallAccuracy(BaseMetric[FunctionCallInput]):
348
+ """
349
+ Comprehensive function call accuracy evaluation.
350
+
351
+ Evaluates:
352
+ - Function name match (weighted 40%)
353
+ - Parameter presence (weighted 30%)
354
+ - Parameter value accuracy (weighted 30%)
355
+
356
+ Returns overall score from 0.0 to 1.0.
357
+ """
358
+
359
+ @property
360
+ def metric_name(self) -> str:
361
+ return "function_call_accuracy"
362
+
363
+ def __init__(self, config: Optional[Dict[str, Any]] = None):
364
+ super().__init__(config)
365
+ self.name_weight = self.config.get("name_weight", 0.4)
366
+ self.presence_weight = self.config.get("presence_weight", 0.3)
367
+ self.value_weight = self.config.get("value_weight", 0.3)
368
+
369
+ def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
370
+ actual_calls = _parse_function_calls(inputs.response)
371
+ expected_calls = _parse_function_calls(inputs.expected_response)
372
+
373
+ if not actual_calls:
374
+ return {
375
+ "output": 0.0,
376
+ "reason": "Could not parse any function calls from response."
377
+ }
378
+
379
+ if not expected_calls:
380
+ return {
381
+ "output": 0.0,
382
+ "reason": "No expected function calls provided."
383
+ }
384
+
385
+ # Handle single vs multiple calls
386
+ if len(expected_calls) == 1 and len(actual_calls) == 1:
387
+ return self._evaluate_single(
388
+ actual_calls[0], expected_calls[0], inputs
389
+ )
390
+
391
+ # Multiple calls - evaluate as set or sequence
392
+ return self._evaluate_multiple(
393
+ actual_calls, expected_calls, inputs
394
+ )
395
+
396
+ def _evaluate_single(
397
+ self,
398
+ actual: FunctionCall,
399
+ expected: FunctionCall,
400
+ inputs: FunctionCallInput
401
+ ) -> Dict[str, Any]:
402
+ """Evaluate a single function call pair."""
403
+ details = []
404
+
405
+ # Name match (40%)
406
+ name_score = 1.0 if actual.name == expected.name else 0.0
407
+ details.append(f"name: {name_score:.0%}")
408
+
409
+ # Parameter presence (30%)
410
+ expected_params = set(expected.arguments.keys())
411
+ actual_params = set(actual.arguments.keys())
412
+
413
+ if expected_params:
414
+ presence_score = len(expected_params & actual_params) / len(expected_params)
415
+ else:
416
+ presence_score = 1.0 if not actual_params or inputs.ignore_extra_params else 0.0
417
+
418
+ details.append(f"params: {presence_score:.0%}")
419
+
420
+ # Parameter values (30%)
421
+ value_matches = 0
422
+ value_total = len(expected.arguments)
423
+
424
+ for param_name, expected_value in expected.arguments.items():
425
+ if param_name in actual.arguments:
426
+ actual_value = actual.arguments[param_name]
427
+ if _values_equal(actual_value, expected_value, inputs.strict_type_check):
428
+ value_matches += 1
429
+
430
+ value_score = value_matches / value_total if value_total > 0 else 1.0
431
+ details.append(f"values: {value_score:.0%}")
432
+
433
+ # Calculate weighted score
434
+ total_score = (
435
+ name_score * self.name_weight +
436
+ presence_score * self.presence_weight +
437
+ value_score * self.value_weight
438
+ )
439
+
440
+ return {
441
+ "output": round(total_score, 4),
442
+ "reason": f"Function call evaluation: {', '.join(details)}. Overall: {total_score:.1%}"
443
+ }
444
+
445
+ def _evaluate_multiple(
446
+ self,
447
+ actual_calls: List[FunctionCall],
448
+ expected_calls: List[FunctionCall],
449
+ inputs: FunctionCallInput
450
+ ) -> Dict[str, Any]:
451
+ """Evaluate multiple function calls (parallel calling)."""
452
+ if inputs.order_matters:
453
+ # Sequence comparison
454
+ if len(actual_calls) != len(expected_calls):
455
+ return {
456
+ "output": 0.0,
457
+ "reason": f"Call count mismatch: got {len(actual_calls)}, expected {len(expected_calls)}"
458
+ }
459
+
460
+ scores = []
461
+ for actual, expected in zip(actual_calls, expected_calls):
462
+ result = self._evaluate_single(actual, expected, inputs)
463
+ scores.append(result["output"])
464
+
465
+ avg_score = sum(scores) / len(scores)
466
+ return {
467
+ "output": round(avg_score, 4),
468
+ "reason": f"Sequence evaluation: {len(scores)} calls, avg score {avg_score:.1%}"
469
+ }
470
+
471
+ # Set comparison - find best match for each expected call
472
+ matched_scores = []
473
+ unmatched_expected = []
474
+
475
+ for expected in expected_calls:
476
+ best_score = 0.0
477
+ for actual in actual_calls:
478
+ result = self._evaluate_single(actual, expected, inputs)
479
+ best_score = max(best_score, result["output"])
480
+
481
+ if best_score > 0:
482
+ matched_scores.append(best_score)
483
+ else:
484
+ unmatched_expected.append(expected.name)
485
+
486
+ if not matched_scores:
487
+ return {
488
+ "output": 0.0,
489
+ "reason": f"No expected calls matched. Expected: {[c.name for c in expected_calls]}"
490
+ }
491
+
492
+ # Penalize for missing calls
493
+ coverage = len(matched_scores) / len(expected_calls)
494
+ avg_match_score = sum(matched_scores) / len(matched_scores)
495
+ final_score = coverage * avg_match_score
496
+
497
+ reason = f"Matched {len(matched_scores)}/{len(expected_calls)} calls, avg accuracy {avg_match_score:.1%}"
498
+ if unmatched_expected:
499
+ reason += f". Missing: {unmatched_expected}"
500
+
501
+ return {
502
+ "output": round(final_score, 4),
503
+ "reason": reason
504
+ }
505
+
506
+
507
+ class FunctionCallExactMatch(BaseMetric[FunctionCallInput]):
508
+ """
509
+ AST-based exact match evaluation.
510
+
511
+ Parses function calls as AST and compares structure.
512
+ Useful for evaluating code-style function calls.
513
+
514
+ Returns 1.0 for exact match, 0.0 otherwise.
515
+ """
516
+
517
+ @property
518
+ def metric_name(self) -> str:
519
+ return "function_call_exact_match"
520
+
521
+ def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
522
+ actual = _parse_function_call(inputs.response)
523
+ expected = _parse_function_call(inputs.expected_response)
524
+
525
+ if actual is None:
526
+ return {
527
+ "output": 0.0,
528
+ "reason": "Could not parse actual function call."
529
+ }
530
+
531
+ if expected is None:
532
+ return {
533
+ "output": 0.0,
534
+ "reason": "Could not parse expected function call."
535
+ }
536
+
537
+ # Compare name
538
+ if actual.name != expected.name:
539
+ return {
540
+ "output": 0.0,
541
+ "reason": f"Function name mismatch: '{actual.name}' vs '{expected.name}'"
542
+ }
543
+
544
+ # Compare arguments
545
+ if set(actual.arguments.keys()) != set(expected.arguments.keys()):
546
+ missing = set(expected.arguments.keys()) - set(actual.arguments.keys())
547
+ extra = set(actual.arguments.keys()) - set(expected.arguments.keys())
548
+ parts = []
549
+ if missing:
550
+ parts.append(f"missing: {missing}")
551
+ if extra and not inputs.ignore_extra_params:
552
+ parts.append(f"extra: {extra}")
553
+ if parts:
554
+ return {
555
+ "output": 0.0,
556
+ "reason": f"Parameter mismatch: {', '.join(parts)}"
557
+ }
558
+
559
+ # Compare values
560
+ for key, expected_value in expected.arguments.items():
561
+ if key not in actual.arguments:
562
+ continue
563
+ actual_value = actual.arguments[key]
564
+ if not _values_equal(actual_value, expected_value, inputs.strict_type_check):
565
+ return {
566
+ "output": 0.0,
567
+ "reason": f"Value mismatch for '{key}': got {actual_value!r}, expected {expected_value!r}"
568
+ }
569
+
570
+ return {
571
+ "output": 1.0,
572
+ "reason": f"Function call matches: {actual.name}({', '.join(f'{k}={v!r}' for k, v in actual.arguments.items())})"
573
+ }
@@ -0,0 +1,87 @@
1
+ """
2
+ Types for Function Calling Evaluation.
3
+
4
+ These types support the evaluation of LLM function/tool calling
5
+ capabilities with AST-based comparison.
6
+ """
7
+
8
+ from typing import Any, Dict, List, Optional, Union
9
+ from pydantic import BaseModel, Field
10
+
11
+ from ...types import BaseMetricInput
12
+
13
+
14
+ class ParameterSpec(BaseModel):
15
+ """Specification for a function parameter."""
16
+
17
+ name: str = Field(..., description="Parameter name")
18
+ type: str = Field(..., description="Expected type (string, integer, number, boolean, array, object)")
19
+ required: bool = Field(default=True, description="Whether the parameter is required")
20
+ enum: Optional[List[Any]] = Field(default=None, description="Allowed values if constrained")
21
+ default: Optional[Any] = Field(default=None, description="Default value if not required")
22
+
23
+
24
+ class FunctionDefinition(BaseModel):
25
+ """Definition of an expected function signature."""
26
+
27
+ name: str = Field(..., description="Function name")
28
+ parameters: List[ParameterSpec] = Field(
29
+ default_factory=list,
30
+ description="List of parameter specifications"
31
+ )
32
+ description: Optional[str] = Field(default=None, description="Function description")
33
+
34
+
35
+ class FunctionCall(BaseModel):
36
+ """Represents a function call from the LLM."""
37
+
38
+ name: str = Field(..., description="Name of the function called")
39
+ arguments: Dict[str, Any] = Field(
40
+ default_factory=dict,
41
+ description="Arguments passed to the function"
42
+ )
43
+
44
+
45
+ class FunctionCallInput(BaseMetricInput):
46
+ """
47
+ Input for function calling evaluation metrics.
48
+
49
+ Supports evaluating:
50
+ - Single function call against expected
51
+ - Multiple function calls (parallel calling)
52
+ - Function call with schema validation
53
+ """
54
+
55
+ # The actual function call(s) from the LLM
56
+ response: Union[FunctionCall, List[FunctionCall], Dict[str, Any], str] = Field(
57
+ ...,
58
+ description="The function call(s) from the LLM. Can be FunctionCall object, dict, JSON string, or list of calls."
59
+ )
60
+
61
+ # The expected function call(s)
62
+ expected_response: Optional[Union[FunctionCall, List[FunctionCall], Dict[str, Any], str]] = Field(
63
+ default=None,
64
+ description="The expected function call(s). Required for accuracy metrics."
65
+ )
66
+
67
+ # Function definitions for schema validation
68
+ function_definitions: Optional[List[FunctionDefinition]] = Field(
69
+ default=None,
70
+ description="Available function definitions for validation."
71
+ )
72
+
73
+ # Evaluation options
74
+ strict_type_check: bool = Field(
75
+ default=False,
76
+ description="If True, require exact type matches. If False, allow compatible types (e.g., int/float)."
77
+ )
78
+
79
+ ignore_extra_params: bool = Field(
80
+ default=False,
81
+ description="If True, ignore extra parameters not in expected. If False, penalize extra params."
82
+ )
83
+
84
+ order_matters: bool = Field(
85
+ default=False,
86
+ description="If True, for parallel calls, order must match. If False, set comparison."
87
+ )