@anvia/core 1.1.1 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (598) hide show
  1. package/README.md +226 -0
  2. package/dist/agent/agent-stream.js +7 -0
  3. package/dist/agent/agent-stream.js.map +1 -0
  4. package/dist/agent/agent-tool.js +24 -0
  5. package/dist/agent/agent-tool.js.map +1 -0
  6. package/dist/agent/agent.js +89 -0
  7. package/dist/agent/agent.js.map +1 -0
  8. package/dist/agent/errors.js +17 -0
  9. package/dist/agent/errors.js.map +1 -0
  10. package/dist/agent/ids.js +7 -0
  11. package/dist/agent/ids.js.map +1 -0
  12. package/dist/agent/index.d.ts +127 -9
  13. package/dist/agent/index.js +103 -21
  14. package/dist/agent/interactions/index.js +2 -2
  15. package/dist/agent/interactions.js +22 -0
  16. package/dist/agent/interactions.js.map +1 -0
  17. package/dist/agent/lifecycle.js +9 -0
  18. package/dist/agent/lifecycle.js.map +1 -0
  19. package/dist/agent/output-schema.js +10 -0
  20. package/dist/agent/output-schema.js.map +1 -0
  21. package/dist/agent/resolve-options.js +46 -0
  22. package/dist/agent/resolve-options.js.map +1 -0
  23. package/dist/agent/resolved-agent.js +92 -0
  24. package/dist/agent/resolved-agent.js.map +1 -0
  25. package/dist/agent/run-types.js +1 -0
  26. package/dist/agent/run-types.js.map +1 -0
  27. package/dist/agent/snapshot.js +17 -0
  28. package/dist/agent/snapshot.js.map +1 -0
  29. package/dist/agent/team/agent-team.js +98 -0
  30. package/dist/agent/team/agent-team.js.map +1 -0
  31. package/dist/agent/team/errors.js +9 -0
  32. package/dist/agent/team/errors.js.map +1 -0
  33. package/dist/agent/team/index.js +104 -0
  34. package/dist/agent/team/index.js.map +1 -0
  35. package/dist/agent/team/types.js +1 -0
  36. package/dist/agent/team/types.js.map +1 -0
  37. package/dist/agent/tool-catalog.js +37 -0
  38. package/dist/agent/tool-catalog.js.map +1 -0
  39. package/dist/agent/tool-state.js +11 -0
  40. package/dist/agent/tool-state.js.map +1 -0
  41. package/dist/agent/types.js +1 -0
  42. package/dist/agent/types.js.map +1 -0
  43. package/dist/agent/vector-context.js +10 -0
  44. package/dist/agent/vector-context.js.map +1 -0
  45. package/dist/{agent-DTmzs1Qr.d.ts → agent-C9VhENdj.d.ts} +2 -3
  46. package/dist/{chunk-2QZJTG2E.js → chunk-23GEYG2X.js} +11 -71
  47. package/dist/chunk-23GEYG2X.js.map +1 -0
  48. package/dist/{chunk-I3Q7XSBB.js → chunk-23LRXKQX.js} +6 -4
  49. package/dist/{chunk-I3Q7XSBB.js.map → chunk-23LRXKQX.js.map} +1 -1
  50. package/dist/{chunk-YK4WAAS4.js → chunk-2NLINXRF.js} +1 -1
  51. package/dist/{chunk-SPD4XVR7.js → chunk-2XSMGVBG.js} +1 -1
  52. package/dist/chunk-3J2CSVPR.js +1 -0
  53. package/dist/chunk-3J2CSVPR.js.map +1 -0
  54. package/dist/chunk-3P3YVJ7V.js +1 -0
  55. package/dist/chunk-3P3YVJ7V.js.map +1 -0
  56. package/dist/chunk-3TAQXJBC.js +107 -0
  57. package/dist/chunk-3TAQXJBC.js.map +1 -0
  58. package/dist/chunk-47NTHIVB.js +76 -0
  59. package/dist/chunk-47NTHIVB.js.map +1 -0
  60. package/dist/chunk-4AW2633B.js +9 -0
  61. package/dist/chunk-4AW2633B.js.map +1 -0
  62. package/dist/{chunk-3XQGVDU5.js → chunk-4C7AXYOM.js} +1 -1
  63. package/dist/{chunk-IZNOP6JG.js → chunk-4FIVU53H.js} +1 -1
  64. package/dist/chunk-4ITAV7YA.js +1 -0
  65. package/dist/chunk-4ITAV7YA.js.map +1 -0
  66. package/dist/{chunk-7JLAIN6E.js → chunk-4ONUNWGC.js} +2 -2
  67. package/dist/chunk-4UUIGAIX.js +1 -0
  68. package/dist/chunk-4UUIGAIX.js.map +1 -0
  69. package/dist/chunk-53TPAXYR.js +1 -0
  70. package/dist/chunk-53TPAXYR.js.map +1 -0
  71. package/dist/chunk-5AETKC4L.js +107 -0
  72. package/dist/chunk-5AETKC4L.js.map +1 -0
  73. package/dist/chunk-5I2NVRHC.js +146 -0
  74. package/dist/chunk-5I2NVRHC.js.map +1 -0
  75. package/dist/chunk-5IURMWFN.js +53 -0
  76. package/dist/chunk-5IURMWFN.js.map +1 -0
  77. package/dist/chunk-5M6GTA3T.js +1 -0
  78. package/dist/chunk-5M6GTA3T.js.map +1 -0
  79. package/dist/chunk-5N5W7CX7.js +18 -0
  80. package/dist/chunk-5N5W7CX7.js.map +1 -0
  81. package/dist/chunk-5P2YOQSA.js +355 -0
  82. package/dist/chunk-5P2YOQSA.js.map +1 -0
  83. package/dist/chunk-5S7T2URX.js +1 -0
  84. package/dist/chunk-5S7T2URX.js.map +1 -0
  85. package/dist/chunk-5V533R7F.js +19 -0
  86. package/dist/chunk-5V533R7F.js.map +1 -0
  87. package/dist/chunk-5WLIHNNJ.js +56 -0
  88. package/dist/chunk-5WLIHNNJ.js.map +1 -0
  89. package/dist/chunk-6BTJC3ES.js +37 -0
  90. package/dist/chunk-6BTJC3ES.js.map +1 -0
  91. package/dist/chunk-6N4ENZO4.js +24 -0
  92. package/dist/chunk-6N4ENZO4.js.map +1 -0
  93. package/dist/{chunk-4UQALF5Y.js → chunk-6RQYWZI3.js} +2 -36
  94. package/dist/chunk-6RQYWZI3.js.map +1 -0
  95. package/dist/chunk-6TAWRKFK.js +86 -0
  96. package/dist/chunk-6TAWRKFK.js.map +1 -0
  97. package/dist/chunk-7F53DOML.js +53 -0
  98. package/dist/chunk-7F53DOML.js.map +1 -0
  99. package/dist/chunk-7KAGWJ5M.js +112 -0
  100. package/dist/chunk-7KAGWJ5M.js.map +1 -0
  101. package/dist/chunk-7LBPRXZP.js +1 -0
  102. package/dist/chunk-7LBPRXZP.js.map +1 -0
  103. package/dist/chunk-A7BXVZXZ.js +25 -0
  104. package/dist/chunk-A7BXVZXZ.js.map +1 -0
  105. package/dist/chunk-B47EXUHL.js +67 -0
  106. package/dist/chunk-B47EXUHL.js.map +1 -0
  107. package/dist/chunk-B5JQDR33.js +14 -0
  108. package/dist/chunk-B5JQDR33.js.map +1 -0
  109. package/dist/chunk-B5YTCKUJ.js +158 -0
  110. package/dist/chunk-B5YTCKUJ.js.map +1 -0
  111. package/dist/chunk-BDDC72GF.js +189 -0
  112. package/dist/chunk-BDDC72GF.js.map +1 -0
  113. package/dist/chunk-BDKFFFVX.js +342 -0
  114. package/dist/chunk-BDKFFFVX.js.map +1 -0
  115. package/dist/{chunk-QGX73TSQ.js → chunk-BONQTXRQ.js} +3 -12
  116. package/dist/chunk-BONQTXRQ.js.map +1 -0
  117. package/dist/chunk-BUPC72Y2.js +1007 -0
  118. package/dist/chunk-BUPC72Y2.js.map +1 -0
  119. package/dist/chunk-CKKNO4XB.js +70 -0
  120. package/dist/chunk-CKKNO4XB.js.map +1 -0
  121. package/dist/chunk-CX7ZVTO5.js +22 -0
  122. package/dist/chunk-CX7ZVTO5.js.map +1 -0
  123. package/dist/chunk-DU4WA6LQ.js +1 -0
  124. package/dist/chunk-DU4WA6LQ.js.map +1 -0
  125. package/dist/chunk-EJR6VVB3.js +63 -0
  126. package/dist/chunk-EJR6VVB3.js.map +1 -0
  127. package/dist/chunk-EKZRAWBP.js +110 -0
  128. package/dist/chunk-EKZRAWBP.js.map +1 -0
  129. package/dist/chunk-ENZ2XX27.js +26 -0
  130. package/dist/chunk-ENZ2XX27.js.map +1 -0
  131. package/dist/chunk-EOAJBBII.js +1 -0
  132. package/dist/chunk-EOAJBBII.js.map +1 -0
  133. package/dist/chunk-ERN7P6Q4.js +1 -0
  134. package/dist/chunk-ERN7P6Q4.js.map +1 -0
  135. package/dist/chunk-ESOFEAYK.js +57 -0
  136. package/dist/chunk-ESOFEAYK.js.map +1 -0
  137. package/dist/chunk-ETYB5JO6.js +175 -0
  138. package/dist/chunk-ETYB5JO6.js.map +1 -0
  139. package/dist/chunk-EZR5AV3J.js +74 -0
  140. package/dist/chunk-EZR5AV3J.js.map +1 -0
  141. package/dist/chunk-F2OKLHGB.js +9 -0
  142. package/dist/chunk-F2OKLHGB.js.map +1 -0
  143. package/dist/chunk-F3E2RLC7.js +342 -0
  144. package/dist/chunk-F3E2RLC7.js.map +1 -0
  145. package/dist/chunk-FAPIZJS2.js +54 -0
  146. package/dist/chunk-FAPIZJS2.js.map +1 -0
  147. package/dist/chunk-GDJNOZ47.js +1 -0
  148. package/dist/chunk-GDJNOZ47.js.map +1 -0
  149. package/dist/chunk-GDY2Z2SZ.js +83 -0
  150. package/dist/chunk-GDY2Z2SZ.js.map +1 -0
  151. package/dist/chunk-GLMYZ6OA.js +20 -0
  152. package/dist/chunk-GLMYZ6OA.js.map +1 -0
  153. package/dist/{chunk-XEVOC433.js → chunk-GPVXCY5I.js} +2 -73
  154. package/dist/chunk-GPVXCY5I.js.map +1 -0
  155. package/dist/{chunk-KSKST3KP.js → chunk-H3H74RWJ.js} +1 -1
  156. package/dist/chunk-HAAGSIZH.js +89 -0
  157. package/dist/chunk-HAAGSIZH.js.map +1 -0
  158. package/dist/{chunk-K5L7R7XM.js → chunk-HHMLE3TZ.js} +6 -54
  159. package/dist/chunk-HHMLE3TZ.js.map +1 -0
  160. package/dist/{chunk-JLJMVRRE.js → chunk-HP34GRQX.js} +10 -83
  161. package/dist/chunk-HP34GRQX.js.map +1 -0
  162. package/dist/chunk-HTI343NP.js +1 -0
  163. package/dist/chunk-HTI343NP.js.map +1 -0
  164. package/dist/{chunk-T2C3CGDQ.js → chunk-HTI7DZJM.js} +6 -4
  165. package/dist/{chunk-T2C3CGDQ.js.map → chunk-HTI7DZJM.js.map} +1 -1
  166. package/dist/chunk-IDNL6IH4.js +294 -0
  167. package/dist/chunk-IDNL6IH4.js.map +1 -0
  168. package/dist/chunk-IV6YQTY2.js +47 -0
  169. package/dist/chunk-IV6YQTY2.js.map +1 -0
  170. package/dist/chunk-J4SINPUA.js +939 -0
  171. package/dist/chunk-J4SINPUA.js.map +1 -0
  172. package/dist/chunk-J6S6MT7O.js +32 -0
  173. package/dist/chunk-J6S6MT7O.js.map +1 -0
  174. package/dist/chunk-JB3RFK6K.js +168 -0
  175. package/dist/chunk-JB3RFK6K.js.map +1 -0
  176. package/dist/chunk-JBQNCUXA.js +34 -0
  177. package/dist/chunk-JBQNCUXA.js.map +1 -0
  178. package/dist/chunk-JMJFK3A4.js +1 -0
  179. package/dist/chunk-JMJFK3A4.js.map +1 -0
  180. package/dist/chunk-JMMZICU7.js +74 -0
  181. package/dist/chunk-JMMZICU7.js.map +1 -0
  182. package/dist/chunk-JOQ2T3QP.js +663 -0
  183. package/dist/chunk-JOQ2T3QP.js.map +1 -0
  184. package/dist/chunk-K4IBS7VB.js +55 -0
  185. package/dist/chunk-K4IBS7VB.js.map +1 -0
  186. package/dist/chunk-KBBQSP4M.js +16 -0
  187. package/dist/chunk-KBBQSP4M.js.map +1 -0
  188. package/dist/chunk-KFAVEL5F.js +68 -0
  189. package/dist/chunk-KFAVEL5F.js.map +1 -0
  190. package/dist/chunk-KITX3HDN.js +17 -0
  191. package/dist/chunk-KITX3HDN.js.map +1 -0
  192. package/dist/chunk-KO5NMTAM.js +1 -0
  193. package/dist/chunk-KO5NMTAM.js.map +1 -0
  194. package/dist/chunk-KY4VXIN2.js +118 -0
  195. package/dist/chunk-KY4VXIN2.js.map +1 -0
  196. package/dist/chunk-KZGSXJKU.js +29 -0
  197. package/dist/chunk-KZGSXJKU.js.map +1 -0
  198. package/dist/chunk-LAIATEUB.js +24 -0
  199. package/dist/chunk-LAIATEUB.js.map +1 -0
  200. package/dist/{chunk-OJBFDBLG.js → chunk-LBYJPMHC.js} +2 -2
  201. package/dist/chunk-LKIABZNR.js +49 -0
  202. package/dist/chunk-LKIABZNR.js.map +1 -0
  203. package/dist/chunk-LQJ4KHMY.js +23 -0
  204. package/dist/chunk-LQJ4KHMY.js.map +1 -0
  205. package/dist/chunk-MEN4AFDZ.js +1 -0
  206. package/dist/chunk-MEN4AFDZ.js.map +1 -0
  207. package/dist/chunk-MHOJJYDK.js +93 -0
  208. package/dist/chunk-MHOJJYDK.js.map +1 -0
  209. package/dist/chunk-MJCMKZFP.js +30 -0
  210. package/dist/chunk-MJCMKZFP.js.map +1 -0
  211. package/dist/chunk-MT3BLVYN.js +31 -0
  212. package/dist/chunk-MT3BLVYN.js.map +1 -0
  213. package/dist/chunk-NF6WBPFG.js +37 -0
  214. package/dist/chunk-NF6WBPFG.js.map +1 -0
  215. package/dist/chunk-NNEKBZSM.js +66 -0
  216. package/dist/chunk-NNEKBZSM.js.map +1 -0
  217. package/dist/chunk-NPJ2T3L5.js +19 -0
  218. package/dist/chunk-NPJ2T3L5.js.map +1 -0
  219. package/dist/chunk-NVALGVQG.js +150 -0
  220. package/dist/chunk-NVALGVQG.js.map +1 -0
  221. package/dist/chunk-OJF67RNM.js +1 -0
  222. package/dist/chunk-OJF67RNM.js.map +1 -0
  223. package/dist/chunk-PHLGGRHC.js +89 -0
  224. package/dist/chunk-PHLGGRHC.js.map +1 -0
  225. package/dist/chunk-PIMD4DYG.js +104 -0
  226. package/dist/chunk-PIMD4DYG.js.map +1 -0
  227. package/dist/chunk-PJNE75WC.js +1 -0
  228. package/dist/chunk-PJNE75WC.js.map +1 -0
  229. package/dist/{chunk-J6LVLV6P.js → chunk-PLDJCCVL.js} +1 -1
  230. package/dist/{chunk-3RWESPUG.js → chunk-POXHJF3H.js} +6 -51
  231. package/dist/chunk-POXHJF3H.js.map +1 -0
  232. package/dist/chunk-PPJ7SYQL.js +81 -0
  233. package/dist/chunk-PPJ7SYQL.js.map +1 -0
  234. package/dist/chunk-PT777EQ3.js +48 -0
  235. package/dist/chunk-PT777EQ3.js.map +1 -0
  236. package/dist/chunk-PYLPLFJZ.js +318 -0
  237. package/dist/chunk-PYLPLFJZ.js.map +1 -0
  238. package/dist/chunk-QEPBHHAP.js +41 -0
  239. package/dist/chunk-QEPBHHAP.js.map +1 -0
  240. package/dist/chunk-QHGYCYV6.js +91 -0
  241. package/dist/chunk-QHGYCYV6.js.map +1 -0
  242. package/dist/chunk-R6722CIU.js +28 -0
  243. package/dist/chunk-R6722CIU.js.map +1 -0
  244. package/dist/chunk-RA4YVN43.js +1 -0
  245. package/dist/chunk-RA4YVN43.js.map +1 -0
  246. package/dist/chunk-RT5LUEO3.js +2339 -0
  247. package/dist/chunk-RT5LUEO3.js.map +1 -0
  248. package/dist/chunk-RXKUIJ77.js +236 -0
  249. package/dist/chunk-RXKUIJ77.js.map +1 -0
  250. package/dist/chunk-SC4SUIEY.js +1 -0
  251. package/dist/chunk-SC4SUIEY.js.map +1 -0
  252. package/dist/chunk-SQAAVYJG.js +12 -0
  253. package/dist/chunk-SQAAVYJG.js.map +1 -0
  254. package/dist/chunk-SXE4J43E.js +59 -0
  255. package/dist/chunk-SXE4J43E.js.map +1 -0
  256. package/dist/{chunk-EFLT7XZD.js → chunk-SYULMZGL.js} +6 -4
  257. package/dist/{chunk-EFLT7XZD.js.map → chunk-SYULMZGL.js.map} +1 -1
  258. package/dist/chunk-T4KS577P.js +58 -0
  259. package/dist/chunk-T4KS577P.js.map +1 -0
  260. package/dist/chunk-TFIOU6UR.js +78 -0
  261. package/dist/chunk-TFIOU6UR.js.map +1 -0
  262. package/dist/chunk-TIZGADU4.js +375 -0
  263. package/dist/chunk-TIZGADU4.js.map +1 -0
  264. package/dist/chunk-TP3MCUXS.js +1 -0
  265. package/dist/chunk-TP3MCUXS.js.map +1 -0
  266. package/dist/chunk-TSPQD5HW.js +613 -0
  267. package/dist/chunk-TSPQD5HW.js.map +1 -0
  268. package/dist/{chunk-JTJU56ZV.js → chunk-UOGM62JL.js} +11 -7
  269. package/dist/{chunk-JTJU56ZV.js.map → chunk-UOGM62JL.js.map} +1 -1
  270. package/dist/chunk-USRKPEQN.js +55 -0
  271. package/dist/chunk-USRKPEQN.js.map +1 -0
  272. package/dist/chunk-UTAB3XQI.js +667 -0
  273. package/dist/chunk-UTAB3XQI.js.map +1 -0
  274. package/dist/chunk-V3FFZWF3.js +182 -0
  275. package/dist/chunk-V3FFZWF3.js.map +1 -0
  276. package/dist/chunk-V4OZ7ISA.js +9 -0
  277. package/dist/chunk-V4OZ7ISA.js.map +1 -0
  278. package/dist/chunk-VZ5ZZBNX.js +139 -0
  279. package/dist/chunk-VZ5ZZBNX.js.map +1 -0
  280. package/dist/{chunk-AHLKV6KP.js → chunk-WN6AVBO4.js} +1 -1
  281. package/dist/chunk-WXR5CWSC.js +1 -0
  282. package/dist/chunk-WXR5CWSC.js.map +1 -0
  283. package/dist/chunk-X6WS3XWL.js +36 -0
  284. package/dist/chunk-X6WS3XWL.js.map +1 -0
  285. package/dist/{chunk-Q5BCNYED.js → chunk-XJRNBTIV.js} +7 -4
  286. package/dist/chunk-XQV3XNVT.js +225 -0
  287. package/dist/chunk-XQV3XNVT.js.map +1 -0
  288. package/dist/chunk-XV3DH5GC.js +58 -0
  289. package/dist/chunk-XV3DH5GC.js.map +1 -0
  290. package/dist/chunk-YWRXOPHC.js +43 -0
  291. package/dist/chunk-YWRXOPHC.js.map +1 -0
  292. package/dist/chunk-YZVFIW5D.js +27 -0
  293. package/dist/chunk-YZVFIW5D.js.map +1 -0
  294. package/dist/chunk-ZHC5CD6J.js +33 -0
  295. package/dist/chunk-ZHC5CD6J.js.map +1 -0
  296. package/dist/chunk-ZPJFXRQZ.js +23 -0
  297. package/dist/chunk-ZPJFXRQZ.js.map +1 -0
  298. package/dist/chunk-ZPXOBQDS.js +57 -0
  299. package/dist/chunk-ZPXOBQDS.js.map +1 -0
  300. package/dist/chunk-ZUFS6R7L.js +155 -0
  301. package/dist/chunk-ZUFS6R7L.js.map +1 -0
  302. package/dist/chunk-ZXLCTQUX.js +62 -0
  303. package/dist/chunk-ZXLCTQUX.js.map +1 -0
  304. package/dist/chunk-ZYXMWEL4.js +59 -0
  305. package/dist/chunk-ZYXMWEL4.js.map +1 -0
  306. package/dist/completion/controls.js +11 -0
  307. package/dist/completion/controls.js.map +1 -0
  308. package/dist/completion/documents.js +9 -0
  309. package/dist/completion/documents.js.map +1 -0
  310. package/dist/completion/generate-completion.js +23 -0
  311. package/dist/completion/generate-completion.js.map +1 -0
  312. package/dist/completion/index.js +23 -12
  313. package/dist/completion/json.js +7 -0
  314. package/dist/completion/json.js.map +1 -0
  315. package/dist/completion/message-schema.js +18 -0
  316. package/dist/completion/message-schema.js.map +1 -0
  317. package/dist/completion/provider-output-error.js +12 -0
  318. package/dist/completion/provider-output-error.js.map +1 -0
  319. package/dist/completion/stream-accumulator.js +10 -0
  320. package/dist/completion/stream-accumulator.js.map +1 -0
  321. package/dist/completion/types.js +28 -0
  322. package/dist/completion/types.js.map +1 -0
  323. package/dist/documents/chunk-text.js +7 -0
  324. package/dist/documents/chunk-text.js.map +1 -0
  325. package/dist/documents/index.js +5 -2
  326. package/dist/documents/text-document.js +8 -0
  327. package/dist/documents/text-document.js.map +1 -0
  328. package/dist/embeddings/distance.js +17 -0
  329. package/dist/embeddings/distance.js.map +1 -0
  330. package/dist/embeddings/embed.js +18 -0
  331. package/dist/embeddings/embed.js.map +1 -0
  332. package/dist/embeddings/index.js +11 -6
  333. package/dist/embeddings/types.js +2 -0
  334. package/dist/embeddings/types.js.map +1 -0
  335. package/dist/evals/advanced-metrics.js +46 -0
  336. package/dist/evals/advanced-metrics.js.map +1 -0
  337. package/dist/evals/agent-target.js +10 -0
  338. package/dist/evals/agent-target.js.map +1 -0
  339. package/dist/evals/cli.js +40 -0
  340. package/dist/evals/cli.js.map +1 -0
  341. package/dist/evals/execution.js +15 -0
  342. package/dist/evals/execution.js.map +1 -0
  343. package/dist/evals/format.js +13 -0
  344. package/dist/evals/format.js.map +1 -0
  345. package/dist/evals/index.d.ts +4 -4
  346. package/dist/evals/index.js +87 -2660
  347. package/dist/evals/index.js.map +1 -1
  348. package/dist/evals/judge.js +27 -0
  349. package/dist/evals/judge.js.map +1 -0
  350. package/dist/evals/metric.js +7 -0
  351. package/dist/evals/metric.js.map +1 -0
  352. package/dist/evals/metrics.js +53 -0
  353. package/dist/evals/metrics.js.map +1 -0
  354. package/dist/evals/outcome.js +7 -0
  355. package/dist/evals/outcome.js.map +1 -0
  356. package/dist/evals/reporting.js +11 -0
  357. package/dist/evals/reporting.js.map +1 -0
  358. package/dist/evals/runner.js +29 -0
  359. package/dist/evals/runner.js.map +1 -0
  360. package/dist/evals/selectors.js +18 -0
  361. package/dist/evals/selectors.js.map +1 -0
  362. package/dist/evals/suite.js +11 -0
  363. package/dist/evals/suite.js.map +1 -0
  364. package/dist/evals/types.js +1 -0
  365. package/dist/evals/types.js.map +1 -0
  366. package/dist/extractor/extractor.js +23 -0
  367. package/dist/extractor/extractor.js.map +1 -0
  368. package/dist/extractor/index.js +16 -7
  369. package/dist/guardrails/actions.js +13 -0
  370. package/dist/guardrails/actions.js.map +1 -0
  371. package/dist/guardrails/index.d.ts +3 -4
  372. package/dist/guardrails/index.js +15 -7
  373. package/dist/guardrails/message.js +9 -0
  374. package/dist/guardrails/message.js.map +1 -0
  375. package/dist/guardrails/policy.js +17 -0
  376. package/dist/guardrails/policy.js.map +1 -0
  377. package/dist/guardrails/runtime.js +11 -0
  378. package/dist/guardrails/runtime.js.map +1 -0
  379. package/dist/guardrails/text.js +9 -0
  380. package/dist/guardrails/text.js.map +1 -0
  381. package/dist/guardrails/types.js +1 -0
  382. package/dist/guardrails/types.js.map +1 -0
  383. package/dist/hooks/control.js +17 -0
  384. package/dist/hooks/control.js.map +1 -0
  385. package/dist/hooks/index.js +19 -0
  386. package/dist/hooks/index.js.map +1 -0
  387. package/dist/hooks/types.js +2 -0
  388. package/dist/hooks/types.js.map +1 -0
  389. package/dist/image-generation/generate-image.js +11 -0
  390. package/dist/image-generation/generate-image.js.map +1 -0
  391. package/dist/image-generation/index.js +7 -4
  392. package/dist/image-generation/types.js +2 -0
  393. package/dist/image-generation/types.js.map +1 -0
  394. package/dist/index.d.ts +5 -7
  395. package/dist/index.js +163 -57
  396. package/dist/internal/abort.js +13 -0
  397. package/dist/internal/abort.js.map +1 -0
  398. package/dist/internal/agent-runtime/agent-run.js +81 -0
  399. package/dist/internal/agent-runtime/agent-run.js.map +1 -0
  400. package/dist/internal/agent-runtime/approval-request.js +1 -0
  401. package/dist/internal/agent-runtime/approval-request.js.map +1 -0
  402. package/dist/internal/agent-runtime/approval-requirement.js +9 -0
  403. package/dist/internal/agent-runtime/approval-requirement.js.map +1 -0
  404. package/dist/internal/agent-runtime/continuation-state.js +26 -0
  405. package/dist/internal/agent-runtime/continuation-state.js.map +1 -0
  406. package/dist/internal/agent-runtime/interaction-suspension.js +25 -0
  407. package/dist/internal/agent-runtime/interaction-suspension.js.map +1 -0
  408. package/dist/internal/agent-runtime/memory-scope.js +8 -0
  409. package/dist/internal/agent-runtime/memory-scope.js.map +1 -0
  410. package/dist/internal/agent-runtime/memory.js +24 -0
  411. package/dist/internal/agent-runtime/memory.js.map +1 -0
  412. package/dist/internal/agent-runtime/prepared-tool-call.js +19 -0
  413. package/dist/internal/agent-runtime/prepared-tool-call.js.map +1 -0
  414. package/dist/internal/agent-runtime/retrieval.js +35 -0
  415. package/dist/internal/agent-runtime/retrieval.js.map +1 -0
  416. package/dist/internal/agent-runtime/run-options.js +9 -0
  417. package/dist/internal/agent-runtime/run-options.js.map +1 -0
  418. package/dist/internal/agent-runtime/run-validation.js +9 -0
  419. package/dist/internal/agent-runtime/run-validation.js.map +1 -0
  420. package/dist/internal/agent-runtime/stream-events.js +11 -0
  421. package/dist/internal/agent-runtime/stream-events.js.map +1 -0
  422. package/dist/internal/agent-runtime/structured-output.js +13 -0
  423. package/dist/internal/agent-runtime/structured-output.js.map +1 -0
  424. package/dist/internal/agent-runtime/tool-execution.js +55 -0
  425. package/dist/internal/agent-runtime/tool-execution.js.map +1 -0
  426. package/dist/internal/agent.d.ts +5 -5
  427. package/dist/internal/agent.js +93 -57
  428. package/dist/internal/agent.js.map +1 -1
  429. package/dist/internal/async-queue.js +7 -0
  430. package/dist/internal/async-queue.js.map +1 -0
  431. package/dist/internal/completion-request.js +11 -0
  432. package/dist/internal/completion-request.js.map +1 -0
  433. package/dist/internal/concurrency.js +7 -0
  434. package/dist/internal/concurrency.js.map +1 -0
  435. package/dist/internal/json-object.js +8 -0
  436. package/dist/internal/json-object.js.map +1 -0
  437. package/dist/internal/rag-text.js +7 -0
  438. package/dist/internal/rag-text.js.map +1 -0
  439. package/dist/internal/record.js +7 -0
  440. package/dist/internal/record.js.map +1 -0
  441. package/dist/internal/team-runtime/coordination.js +12 -0
  442. package/dist/internal/team-runtime/coordination.js.map +1 -0
  443. package/dist/internal/team-runtime/member.js +26 -0
  444. package/dist/internal/team-runtime/member.js.map +1 -0
  445. package/dist/internal/team-runtime/policy.js +7 -0
  446. package/dist/internal/team-runtime/policy.js.map +1 -0
  447. package/dist/internal/team-runtime/run.js +95 -0
  448. package/dist/internal/team-runtime/run.js.map +1 -0
  449. package/dist/internal/team-runtime/stream.js +9 -0
  450. package/dist/internal/team-runtime/stream.js.map +1 -0
  451. package/dist/internal/team-runtime/tools.js +17 -0
  452. package/dist/internal/team-runtime/tools.js.map +1 -0
  453. package/dist/internal/type-utils.js +1 -0
  454. package/dist/internal/type-utils.js.map +1 -0
  455. package/dist/internal/vector-search-options.js +9 -0
  456. package/dist/internal/vector-search-options.js.map +1 -0
  457. package/dist/mcp/index.js +3 -1
  458. package/dist/mcp/tool.js +7 -0
  459. package/dist/mcp/tool.js.map +1 -0
  460. package/dist/mcp/types.js +2 -0
  461. package/dist/mcp/types.js.map +1 -0
  462. package/dist/memory/assert.js +11 -0
  463. package/dist/memory/assert.js.map +1 -0
  464. package/dist/memory/compaction.js +26 -0
  465. package/dist/memory/compaction.js.map +1 -0
  466. package/dist/memory/errors.js +9 -0
  467. package/dist/memory/errors.js.map +1 -0
  468. package/dist/memory/index.js +23 -10
  469. package/dist/memory/options.js +10 -0
  470. package/dist/memory/options.js.map +1 -0
  471. package/dist/memory/scope-key.js +7 -0
  472. package/dist/memory/scope-key.js.map +1 -0
  473. package/dist/memory/types.js +2 -0
  474. package/dist/memory/types.js.map +1 -0
  475. package/dist/{types-CRda8x7p.d.ts → middleware-CWjnMbiH.d.ts} +64 -5
  476. package/dist/model-call-options.js +1 -0
  477. package/dist/model-call-options.js.map +1 -0
  478. package/dist/model-listing/errors.js +7 -0
  479. package/dist/model-listing/errors.js.map +1 -0
  480. package/dist/model-listing/index.js +4 -13
  481. package/dist/model-listing/index.js.map +1 -1
  482. package/dist/model-listing/types.js +2 -0
  483. package/dist/model-listing/types.js.map +1 -0
  484. package/dist/observability/group.js +30 -0
  485. package/dist/observability/group.js.map +1 -0
  486. package/dist/observability/index.d.ts +4 -6
  487. package/dist/observability/index.js +16 -8
  488. package/dist/observability/snapshot.js +7 -0
  489. package/dist/observability/snapshot.js.map +1 -0
  490. package/dist/observability/types.js +1 -0
  491. package/dist/observability/types.js.map +1 -0
  492. package/dist/pipeline/errors.js +9 -0
  493. package/dist/pipeline/errors.js.map +1 -0
  494. package/dist/pipeline/graph.js +21 -0
  495. package/dist/pipeline/graph.js.map +1 -0
  496. package/dist/pipeline/index.d.ts +4 -6
  497. package/dist/pipeline/index.js +27 -805
  498. package/dist/pipeline/index.js.map +1 -1
  499. package/dist/pipeline/observability.js +15 -0
  500. package/dist/pipeline/observability.js.map +1 -0
  501. package/dist/pipeline/pipeline.js +30 -0
  502. package/dist/pipeline/pipeline.js.map +1 -0
  503. package/dist/pipeline/runtime.js +17 -0
  504. package/dist/pipeline/runtime.js.map +1 -0
  505. package/dist/pipeline/types.js +1 -0
  506. package/dist/pipeline/types.js.map +1 -0
  507. package/dist/retry.js +20 -0
  508. package/dist/retry.js.map +1 -0
  509. package/dist/schema/index.js +1 -0
  510. package/dist/schema/index.js.map +1 -0
  511. package/dist/schema/zod-schema.js +7 -0
  512. package/dist/schema/zod-schema.js.map +1 -0
  513. package/dist/skills/index.js +41 -12
  514. package/dist/skills/instructions.js +7 -0
  515. package/dist/skills/instructions.js.map +1 -0
  516. package/dist/skills/load.js +39 -0
  517. package/dist/skills/load.js.map +1 -0
  518. package/dist/skills/local.js +9 -0
  519. package/dist/skills/local.js.map +1 -0
  520. package/dist/skills/tools.js +37 -0
  521. package/dist/skills/tools.js.map +1 -0
  522. package/dist/skills/types.js +7 -0
  523. package/dist/skills/types.js.map +1 -0
  524. package/dist/speech-generation/generate-speech.js +11 -0
  525. package/dist/speech-generation/generate-speech.js.map +1 -0
  526. package/dist/speech-generation/index.js +7 -4
  527. package/dist/speech-generation/types.js +2 -0
  528. package/dist/speech-generation/types.js.map +1 -0
  529. package/dist/streaming/index.js +3 -37
  530. package/dist/streaming/index.js.map +1 -1
  531. package/dist/streaming/readable-stream.js +7 -0
  532. package/dist/streaming/readable-stream.js.map +1 -0
  533. package/dist/{text-CmjXxakS.d.ts → text-GbW4WYMg.d.ts} +1 -1
  534. package/dist/tool/create-tool.js +14 -0
  535. package/dist/tool/create-tool.js.map +1 -0
  536. package/dist/tool/dynamic-tools.js +35 -0
  537. package/dist/tool/dynamic-tools.js.map +1 -0
  538. package/dist/tool/errors.js +11 -0
  539. package/dist/tool/errors.js.map +1 -0
  540. package/dist/tool/index.d.ts +3 -2
  541. package/dist/tool/index.js +41 -15
  542. package/dist/tool/middleware.js +7 -0
  543. package/dist/tool/middleware.js.map +1 -0
  544. package/dist/tool/question-tool.js +17 -0
  545. package/dist/tool/question-tool.js.map +1 -0
  546. package/dist/tool/skill-tool-marker.js +9 -0
  547. package/dist/tool/skill-tool-marker.js.map +1 -0
  548. package/dist/tool/think-tool.js +15 -0
  549. package/dist/tool/think-tool.js.map +1 -0
  550. package/dist/tool/tool.js +17 -0
  551. package/dist/tool/tool.js.map +1 -0
  552. package/dist/transcription/index.js +7 -4
  553. package/dist/transcription/transcribe.js +11 -0
  554. package/dist/transcription/transcribe.js.map +1 -0
  555. package/dist/transcription/types.js +2 -0
  556. package/dist/transcription/types.js.map +1 -0
  557. package/dist/{types-h0EL6Xn0.d.ts → types-Bqw0AuFl.d.ts} +6 -2
  558. package/dist/vector-store/filter.js +9 -0
  559. package/dist/vector-store/filter.js.map +1 -0
  560. package/dist/vector-store/index.js +31 -11
  561. package/dist/vector-store/ingest.js +19 -0
  562. package/dist/vector-store/ingest.js.map +1 -0
  563. package/dist/vector-store/lsh.js +7 -0
  564. package/dist/vector-store/lsh.js.map +1 -0
  565. package/dist/vector-store/retrieve.js +15 -0
  566. package/dist/vector-store/retrieve.js.map +1 -0
  567. package/dist/vector-store/types.js +1 -0
  568. package/dist/vector-store/types.js.map +1 -0
  569. package/package.json +8 -5
  570. package/dist/chunk-2QZJTG2E.js.map +0 -1
  571. package/dist/chunk-3RWESPUG.js.map +0 -1
  572. package/dist/chunk-4UQALF5Y.js.map +0 -1
  573. package/dist/chunk-AZB6N7P4.js +0 -320
  574. package/dist/chunk-AZB6N7P4.js.map +0 -1
  575. package/dist/chunk-JLJMVRRE.js.map +0 -1
  576. package/dist/chunk-K5L7R7XM.js.map +0 -1
  577. package/dist/chunk-OI3LSMJG.js +0 -1545
  578. package/dist/chunk-OI3LSMJG.js.map +0 -1
  579. package/dist/chunk-QGX73TSQ.js.map +0 -1
  580. package/dist/chunk-XC3LVC4K.js +0 -753
  581. package/dist/chunk-XC3LVC4K.js.map +0 -1
  582. package/dist/chunk-XEVOC433.js.map +0 -1
  583. package/dist/chunk-XSV5C4R5.js +0 -400
  584. package/dist/chunk-XSV5C4R5.js.map +0 -1
  585. package/dist/chunk-XU54PLSD.js +0 -4786
  586. package/dist/chunk-XU54PLSD.js.map +0 -1
  587. package/dist/middleware-mY_lpjDp.d.ts +0 -59
  588. package/dist/type-utils-CtHVDRn_.d.ts +0 -6
  589. /package/dist/{chunk-YK4WAAS4.js.map → chunk-2NLINXRF.js.map} +0 -0
  590. /package/dist/{chunk-SPD4XVR7.js.map → chunk-2XSMGVBG.js.map} +0 -0
  591. /package/dist/{chunk-3XQGVDU5.js.map → chunk-4C7AXYOM.js.map} +0 -0
  592. /package/dist/{chunk-IZNOP6JG.js.map → chunk-4FIVU53H.js.map} +0 -0
  593. /package/dist/{chunk-7JLAIN6E.js.map → chunk-4ONUNWGC.js.map} +0 -0
  594. /package/dist/{chunk-KSKST3KP.js.map → chunk-H3H74RWJ.js.map} +0 -0
  595. /package/dist/{chunk-OJBFDBLG.js.map → chunk-LBYJPMHC.js.map} +0 -0
  596. /package/dist/{chunk-J6LVLV6P.js.map → chunk-PLDJCCVL.js.map} +0 -0
  597. /package/dist/{chunk-AHLKV6KP.js.map → chunk-WN6AVBO4.js.map} +0 -0
  598. /package/dist/{chunk-Q5BCNYED.js.map → chunk-XJRNBTIV.js.map} +0 -0
@@ -0,0 +1,1007 @@
1
+ import {
2
+ addUsage,
3
+ evaluationMetadata,
4
+ runJudge
5
+ } from "./chunk-LKIABZNR.js";
6
+ import {
7
+ resolveActualText,
8
+ resolveExpected
9
+ } from "./chunk-ZPXOBQDS.js";
10
+ import {
11
+ errorMessage,
12
+ formatValue
13
+ } from "./chunk-ZYXMWEL4.js";
14
+ import {
15
+ EvalOutcome
16
+ } from "./chunk-EJR6VVB3.js";
17
+ import {
18
+ mapWithConcurrency
19
+ } from "./chunk-4FIVU53H.js";
20
+ import {
21
+ Usage
22
+ } from "./chunk-IDNL6IH4.js";
23
+ import {
24
+ isJsonValue
25
+ } from "./chunk-4C7AXYOM.js";
26
+
27
+ // src/evals/advanced-metrics.ts
28
+ import { z } from "zod";
29
+ var statementsSchema = z.object({ statements: z.array(z.string()) });
30
+ var factsSchema = z.object({ facts: z.array(z.string()) });
31
+ var questionsSchema = z.object({ questions: z.array(z.string()) });
32
+ var answersSchema = z.object({ answers: z.array(z.enum(["yes", "no"])) });
33
+ var verdictsSchema = z.object({
34
+ verdicts: z.array(
35
+ z.object({
36
+ verdict: z.enum(["yes", "no", "idk"]),
37
+ reason: z.string().optional()
38
+ })
39
+ )
40
+ });
41
+ var binaryVerdictsSchema = z.object({
42
+ verdicts: z.array(
43
+ z.object({
44
+ verdict: z.enum(["yes", "no"]),
45
+ reason: z.string()
46
+ })
47
+ )
48
+ });
49
+ var binaryVerdictSchema = z.object({
50
+ verdict: z.enum(["yes", "no"]),
51
+ reason: z.string()
52
+ });
53
+ var reasonSchema = z.object({ reason: z.string() });
54
+ var abstentionJudgmentSchema = z.object({
55
+ behavior: z.enum(["abstention", "confident_answer"]),
56
+ grounded: z.boolean(),
57
+ reason: z.string()
58
+ });
59
+ function answerRelevancy(options) {
60
+ const config = metricConfig(options, "answer_relevancy");
61
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
62
+ try {
63
+ const input = await resolveInput(options.input, args);
64
+ const actual = await resolveActualText(options.actual, args);
65
+ const statementResult = await runJudge({
66
+ model: options.model,
67
+ schema: statementsSchema,
68
+ instructions: "Break the answer into concise, independently assessable statements. Return every substantive statement using the schema.",
69
+ prompt: `Answer:
70
+ ${actual}`,
71
+ retries: config.retries
72
+ });
73
+ const statements = statementResult.data.statements;
74
+ let verdicts = [];
75
+ let usage = statementResult.usage;
76
+ if (statements.length > 0) {
77
+ const verdictResult = await runJudge({
78
+ model: options.model,
79
+ schema: verdictsSchema,
80
+ instructions: "Classify each answer statement for relevance to the user input. Use yes for relevant, no for irrelevant, and idk only when relevance is genuinely indeterminate. Preserve order and return one verdict per statement.",
81
+ prompt: jsonPrompt({ input, statements }),
82
+ retries: config.retries
83
+ });
84
+ verdicts = verdictResult.data.verdicts;
85
+ assertSameLength("answer relevancy verdicts", statements, verdicts);
86
+ usage = addUsage(usage, verdictResult.usage);
87
+ }
88
+ const score = verdicts.length === 0 ? 1 : verdicts.filter((verdict) => verdict.verdict !== "no").length / verdicts.length;
89
+ const serializedVerdicts = serializeVerdicts(verdicts);
90
+ const reasonResult = await maybeReason({
91
+ model: options.model,
92
+ includeReason: config.includeReason,
93
+ retries: config.retries,
94
+ metric: "answer relevancy",
95
+ score,
96
+ evidence: { input, verdicts: serializedVerdicts }
97
+ });
98
+ usage = addUsage(usage, reasonResult.usage);
99
+ return higherOutcome({
100
+ score,
101
+ threshold: config.threshold,
102
+ strictMode: config.strictMode,
103
+ comment: reasonResult.reason,
104
+ details: { statements, verdicts: serializedVerdicts },
105
+ usage
106
+ });
107
+ } catch (error) {
108
+ return EvalOutcome.fromError(error);
109
+ }
110
+ });
111
+ }
112
+ function promptAlignment(options) {
113
+ if (options.promptInstructions.length === 0) {
114
+ throw new TypeError("promptAlignment requires at least one prompt instruction.");
115
+ }
116
+ const config = metricConfig(options, "prompt_alignment");
117
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
118
+ try {
119
+ const input = await resolveInput(options.input, args);
120
+ const actual = await resolveActualText(options.actual, args);
121
+ const verdictResult = await runJudge({
122
+ model: options.model,
123
+ schema: binaryVerdictsSchema,
124
+ instructions: "Determine whether the answer follows each prompt instruction. Preserve order and return exactly one yes or no verdict per instruction.",
125
+ prompt: jsonPrompt({ input, actual, instructions: options.promptInstructions }),
126
+ retries: config.retries
127
+ });
128
+ const verdicts = verdictResult.data.verdicts;
129
+ assertSameLength("prompt alignment verdicts", options.promptInstructions, verdicts);
130
+ const score = verdicts.filter((verdict) => verdict.verdict === "yes").length / verdicts.length;
131
+ const reasonResult = await maybeReason({
132
+ model: options.model,
133
+ includeReason: config.includeReason,
134
+ retries: config.retries,
135
+ metric: "prompt alignment",
136
+ score,
137
+ evidence: { verdicts }
138
+ });
139
+ const usage = addUsage(verdictResult.usage, reasonResult.usage);
140
+ return higherOutcome({
141
+ score,
142
+ threshold: config.threshold,
143
+ strictMode: config.strictMode,
144
+ comment: reasonResult.reason,
145
+ details: { promptInstructions: options.promptInstructions, verdicts },
146
+ usage
147
+ });
148
+ } catch (error) {
149
+ return EvalOutcome.fromError(error);
150
+ }
151
+ });
152
+ }
153
+ function jsonCorrectness(options) {
154
+ const threshold = validateThreshold(options.threshold ?? 0.5);
155
+ const retries = validateRetries(options.retries ?? 0);
156
+ const includeReason = options.includeReason ?? true;
157
+ const strictMode = options.strictMode ?? true;
158
+ return numericMetric(
159
+ options.name ?? "json_correctness",
160
+ {
161
+ threshold: strictMode ? 1 : threshold,
162
+ required: options.required ?? true
163
+ },
164
+ "higher_is_better",
165
+ async (args) => {
166
+ try {
167
+ const actual = await resolveActualText(options.actual, args);
168
+ let parsed;
169
+ let validationError;
170
+ try {
171
+ parsed = JSON.parse(actual);
172
+ if (!isJsonValue(parsed)) {
173
+ throw new TypeError("Generated output is not a JSON value.");
174
+ }
175
+ const result = options.schema.safeParse(parsed);
176
+ if (!result.success) {
177
+ validationError = z.prettifyError(result.error);
178
+ }
179
+ } catch (error) {
180
+ validationError = errorMessage(error);
181
+ }
182
+ const score = validationError === void 0 ? 1 : 0;
183
+ let comment;
184
+ let usage = Usage.empty();
185
+ if (includeReason) {
186
+ if (score === 1) {
187
+ comment = "The generated JSON is syntactically valid and matches the expected schema.";
188
+ } else if (options.model === void 0) {
189
+ comment = validationError;
190
+ } else {
191
+ const reasonResult = await runJudge({
192
+ model: options.model,
193
+ schema: reasonSchema,
194
+ instructions: "Briefly explain why the generated JSON does not match the expected schema. Focus on actionable syntax, field, and type problems.",
195
+ prompt: jsonPrompt({ actual, validationError }),
196
+ retries
197
+ });
198
+ comment = reasonResult.data.reason;
199
+ usage = reasonResult.usage;
200
+ }
201
+ }
202
+ return higherOutcome({
203
+ score,
204
+ threshold,
205
+ strictMode,
206
+ comment,
207
+ details: validationError === void 0 ? {} : { validationError },
208
+ usage
209
+ });
210
+ } catch (error) {
211
+ return EvalOutcome.fromError(error);
212
+ }
213
+ }
214
+ );
215
+ }
216
+ function hallucination(options) {
217
+ const config = metricConfig(options, "hallucination");
218
+ return numericMetric(config.name, config, "lower_is_better", async (args) => {
219
+ try {
220
+ const actual = await resolveActualText(options.actual, args);
221
+ const context = await resolveStringList(options.context, args.case.context, args, "context");
222
+ const verdictResult = await runJudge({
223
+ model: options.model,
224
+ schema: binaryVerdictsSchema,
225
+ instructions: "Compare the answer with each trusted context. Use yes when the answer is factually aligned with that context and no when it contradicts it. Preserve order and return one verdict per context.",
226
+ prompt: jsonPrompt({ actual, context }),
227
+ retries: config.retries
228
+ });
229
+ const verdicts = verdictResult.data.verdicts;
230
+ assertSameLength("hallucination verdicts", context, verdicts);
231
+ const score = verdicts.filter((verdict) => verdict.verdict === "no").length / verdicts.length;
232
+ const reasonResult = await maybeReason({
233
+ model: options.model,
234
+ includeReason: config.includeReason,
235
+ retries: config.retries,
236
+ metric: "hallucination",
237
+ score,
238
+ evidence: { verdicts }
239
+ });
240
+ const usage = addUsage(verdictResult.usage, reasonResult.usage);
241
+ return lowerOutcome({
242
+ score,
243
+ threshold: config.threshold,
244
+ strictMode: config.strictMode,
245
+ comment: reasonResult.reason,
246
+ details: { verdicts },
247
+ usage
248
+ });
249
+ } catch (error) {
250
+ return EvalOutcome.fromError(error);
251
+ }
252
+ });
253
+ }
254
+ function faithfulness(options) {
255
+ const config = metricConfig(options, "faithfulness");
256
+ const truthsExtractionLimit = validateOptionalNonNegativeInteger(
257
+ options.truthsExtractionLimit,
258
+ "truthsExtractionLimit"
259
+ );
260
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
261
+ try {
262
+ const actual = await resolveActualText(options.actual, args);
263
+ const retrievalContext = await resolveStringList(
264
+ options.retrievalContext,
265
+ args.case.retrievalContext,
266
+ args,
267
+ "retrievalContext"
268
+ );
269
+ const [truthResult, claimResult] = await Promise.all([
270
+ runJudge({
271
+ model: options.model,
272
+ schema: factsSchema,
273
+ instructions: truthsExtractionInstructions(truthsExtractionLimit),
274
+ prompt: jsonPrompt({ retrievalContext }),
275
+ retries: config.retries
276
+ }),
277
+ runJudge({
278
+ model: options.model,
279
+ schema: factsSchema,
280
+ instructions: "Extract every concise factual claim made by the answer. Return claims in the facts array and omit opinions or purely stylistic text.",
281
+ prompt: `Answer:
282
+ ${actual}`,
283
+ retries: config.retries
284
+ })
285
+ ]);
286
+ const truths = limitValues(truthResult.data.facts, truthsExtractionLimit);
287
+ const claims = claimResult.data.facts;
288
+ let verdicts = [];
289
+ let usage = addUsage(truthResult.usage, claimResult.usage);
290
+ if (claims.length > 0) {
291
+ const verdictResult = await runJudge({
292
+ model: options.model,
293
+ schema: verdictsSchema,
294
+ instructions: "Determine whether each answer claim is supported by the supplied truths. Use yes for supported, no for contradicted or unsupported, and idk for genuinely ambiguous support. Preserve order and return one verdict per claim.",
295
+ prompt: jsonPrompt({ truths, claims }),
296
+ retries: config.retries
297
+ });
298
+ verdicts = verdictResult.data.verdicts;
299
+ assertSameLength("faithfulness verdicts", claims, verdicts);
300
+ usage = addUsage(usage, verdictResult.usage);
301
+ }
302
+ const penalizeAmbiguousClaims = options.penalizeAmbiguousClaims ?? false;
303
+ const supported = verdicts.filter(
304
+ (verdict) => verdict.verdict === "yes" || verdict.verdict === "idk" && !penalizeAmbiguousClaims
305
+ ).length;
306
+ const score = verdicts.length === 0 ? 1 : supported / verdicts.length;
307
+ const serializedVerdicts = serializeVerdicts(verdicts);
308
+ const reasonResult = await maybeReason({
309
+ model: options.model,
310
+ includeReason: config.includeReason,
311
+ retries: config.retries,
312
+ metric: "faithfulness",
313
+ score,
314
+ evidence: { verdicts: serializedVerdicts, penalizeAmbiguousClaims }
315
+ });
316
+ usage = addUsage(usage, reasonResult.usage);
317
+ return higherOutcome({
318
+ score,
319
+ threshold: config.threshold,
320
+ strictMode: config.strictMode,
321
+ comment: reasonResult.reason,
322
+ details: { truths, claims, verdicts: serializedVerdicts, penalizeAmbiguousClaims },
323
+ usage
324
+ });
325
+ } catch (error) {
326
+ return EvalOutcome.fromError(error);
327
+ }
328
+ });
329
+ }
330
+ function abstention(options) {
331
+ const retries = validateRetries(options.retries ?? 0);
332
+ return {
333
+ name: options.name ?? "abstention",
334
+ required: options.required ?? true,
335
+ dataType: "CATEGORICAL",
336
+ async evaluate(args) {
337
+ try {
338
+ const actual = await resolveActualText(options.actual, args);
339
+ const shouldAbstain = await resolveExpected(options.shouldAbstain, args);
340
+ if (typeof shouldAbstain !== "boolean") {
341
+ return EvalOutcome.invalid("abstention shouldAbstain must resolve to a boolean.");
342
+ }
343
+ const context = await resolveAbstentionContext(options.context, args);
344
+ if (!shouldAbstain && context.length === 0) {
345
+ return EvalOutcome.invalid(
346
+ "abstention context must be non-empty when shouldAbstain is false."
347
+ );
348
+ }
349
+ const judgment = await runJudge({
350
+ model: options.model,
351
+ schema: abstentionJudgmentSchema,
352
+ instructions: "Classify whether the answer abstains or gives a confident answer. For a confident answer, grounded is true only when every substantive factual claim is supported by the supplied context. For an abstention, set grounded to false. Return a concise evidence-based reason.",
353
+ prompt: jsonPrompt({ actual, context }),
354
+ retries
355
+ });
356
+ const category = abstentionCategory(
357
+ shouldAbstain,
358
+ judgment.data.behavior,
359
+ judgment.data.grounded
360
+ );
361
+ const outcomeOptions = {
362
+ comment: options.includeReason === false ? void 0 : judgment.data.reason,
363
+ metadata: evaluationMetadata(
364
+ {
365
+ behavior: judgment.data.behavior,
366
+ grounded: judgment.data.grounded,
367
+ shouldAbstain
368
+ },
369
+ judgment.usage
370
+ ),
371
+ usage: judgment.usage
372
+ };
373
+ return category === "correct_abstention" || category === "correct_grounded_answer" ? EvalOutcome.pass(category, outcomeOptions) : EvalOutcome.fail(category, outcomeOptions);
374
+ } catch (error) {
375
+ return EvalOutcome.fromError(error);
376
+ }
377
+ }
378
+ };
379
+ }
380
+ async function resolveAbstentionContext(selectorOrValue, args) {
381
+ const context = selectorOrValue === void 0 ? args.case.retrievalContext ?? [] : typeof selectorOrValue === "function" ? await selectorOrValue(args) : selectorOrValue;
382
+ if (!Array.isArray(context) || context.some((value) => typeof value !== "string")) {
383
+ throw new TypeError("abstention context must be an array of strings.");
384
+ }
385
+ return context;
386
+ }
387
+ function abstentionCategory(shouldAbstain, behavior, grounded) {
388
+ if (behavior === "abstention") {
389
+ return shouldAbstain ? "correct_abstention" : "unnecessary_abstention";
390
+ }
391
+ if (shouldAbstain || !grounded) return "unsupported_confident_answer";
392
+ return "correct_grounded_answer";
393
+ }
394
+ function summarization(options) {
395
+ const config = metricConfig(options, "summarization");
396
+ const questionCount = validatePositiveInteger(options.questionCount ?? 5, "questionCount");
397
+ const truthsExtractionLimit = validateOptionalNonNegativeInteger(
398
+ options.truthsExtractionLimit,
399
+ "truthsExtractionLimit"
400
+ );
401
+ const suppliedQuestions = options.assessmentQuestions !== void 0 && options.assessmentQuestions.length > 0 ? [...options.assessmentQuestions] : void 0;
402
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
403
+ try {
404
+ const input = await resolveInput(options.input, args);
405
+ const actual = await resolveActualText(options.actual, args);
406
+ const questionPromise = suppliedQuestions === void 0 ? runJudge({
407
+ model: options.model,
408
+ schema: questionsSchema,
409
+ instructions: `Generate exactly ${questionCount} important yes-or-no assessment questions whose answers capture the source text's essential information.`,
410
+ prompt: `Source text:
411
+ ${input}`,
412
+ retries: config.retries
413
+ }) : Promise.resolve({
414
+ data: { questions: suppliedQuestions },
415
+ usage: Usage.empty()
416
+ });
417
+ const [truthResult, claimResult, questionResult] = await Promise.all([
418
+ runJudge({
419
+ model: options.model,
420
+ schema: factsSchema,
421
+ instructions: truthsExtractionInstructions(truthsExtractionLimit),
422
+ prompt: `Source text:
423
+ ${input}`,
424
+ retries: config.retries
425
+ }),
426
+ runJudge({
427
+ model: options.model,
428
+ schema: factsSchema,
429
+ instructions: "Extract every concise factual claim made by the summary. Return claims in the facts array.",
430
+ prompt: `Summary:
431
+ ${actual}`,
432
+ retries: config.retries
433
+ }),
434
+ questionPromise
435
+ ]);
436
+ const truths = limitValues(truthResult.data.facts, truthsExtractionLimit);
437
+ const claims = claimResult.data.facts;
438
+ const questions = questionResult.data.questions;
439
+ if (questions.length === 0) {
440
+ throw new Error("Summarization assessment questions must not be empty.");
441
+ }
442
+ const sourceAnswerPromise = runJudge({
443
+ model: options.model,
444
+ schema: answersSchema,
445
+ instructions: "Answer each assessment question using only the supplied text. Return one yes or no answer per question in the same order.",
446
+ prompt: jsonPrompt({ questions, text: input }),
447
+ retries: config.retries
448
+ });
449
+ const summaryAnswerPromise = runJudge({
450
+ model: options.model,
451
+ schema: answersSchema,
452
+ instructions: "Answer each assessment question using only the supplied text. Return one yes or no answer per question in the same order.",
453
+ prompt: jsonPrompt({ questions, text: actual }),
454
+ retries: config.retries
455
+ });
456
+ const alignmentPromise = claims.length === 0 ? Promise.resolve({ data: { verdicts: [] }, usage: Usage.empty() }) : runJudge({
457
+ model: options.model,
458
+ schema: verdictsSchema,
459
+ instructions: "Determine whether each summary claim is supported by the source truths. Use yes for supported, no for contradicted, and idk for unsupported filler or ambiguity. Preserve order.",
460
+ prompt: jsonPrompt({ truths, claims }),
461
+ retries: config.retries
462
+ });
463
+ const [sourceAnswerResult, summaryAnswerResult, alignmentResult] = await Promise.all([
464
+ sourceAnswerPromise,
465
+ summaryAnswerPromise,
466
+ alignmentPromise
467
+ ]);
468
+ assertSameLength("source assessment answers", questions, sourceAnswerResult.data.answers);
469
+ assertSameLength("summary assessment answers", questions, summaryAnswerResult.data.answers);
470
+ assertSameLength("summarization alignment verdicts", claims, alignmentResult.data.verdicts);
471
+ const alignmentVerdicts = alignmentResult.data.verdicts;
472
+ const alignmentScore = alignmentVerdicts.length === 0 ? 0 : alignmentVerdicts.filter((verdict) => verdict.verdict === "yes").length / alignmentVerdicts.length;
473
+ let coverageTotal = 0;
474
+ let coverageMatched = 0;
475
+ const coverageVerdicts = questions.map((question, index) => {
476
+ const originalVerdict = sourceAnswerResult.data.answers[index];
477
+ const summaryVerdict = summaryAnswerResult.data.answers[index];
478
+ if (originalVerdict === "yes") {
479
+ coverageTotal += 1;
480
+ if (summaryVerdict === "yes") coverageMatched += 1;
481
+ }
482
+ return { question, originalVerdict, summaryVerdict };
483
+ });
484
+ const coverageScore = coverageTotal === 0 ? 0 : coverageMatched / coverageTotal;
485
+ const score = Math.min(alignmentScore, coverageScore);
486
+ const serializedAlignmentVerdicts = serializeVerdicts(alignmentVerdicts);
487
+ const reasonResult = await maybeReason({
488
+ model: options.model,
489
+ includeReason: config.includeReason,
490
+ retries: config.retries,
491
+ metric: "summarization",
492
+ score,
493
+ evidence: {
494
+ alignmentVerdicts: serializedAlignmentVerdicts,
495
+ coverageVerdicts,
496
+ alignmentScore,
497
+ coverageScore
498
+ }
499
+ });
500
+ const usage = addUsage(
501
+ truthResult.usage,
502
+ claimResult.usage,
503
+ questionResult.usage,
504
+ sourceAnswerResult.usage,
505
+ summaryAnswerResult.usage,
506
+ alignmentResult.usage,
507
+ reasonResult.usage
508
+ );
509
+ return higherOutcome({
510
+ score,
511
+ threshold: config.threshold,
512
+ strictMode: config.strictMode,
513
+ comment: reasonResult.reason,
514
+ details: {
515
+ truths,
516
+ claims,
517
+ assessmentQuestions: questions,
518
+ alignmentVerdicts: serializedAlignmentVerdicts,
519
+ coverageVerdicts,
520
+ scoreBreakdown: { alignment: alignmentScore, coverage: coverageScore }
521
+ },
522
+ usage
523
+ });
524
+ } catch (error) {
525
+ return EvalOutcome.fromError(error);
526
+ }
527
+ });
528
+ }
529
+ function gEval(options) {
530
+ if (options.name.trim().length === 0) throw new TypeError("gEval name must not be empty.");
531
+ if (options.evaluationParams.length === 0) {
532
+ throw new TypeError("gEval requires at least one evaluation parameter.");
533
+ }
534
+ if (options.criteria === void 0 === (options.evaluationSteps === void 0)) {
535
+ throw new TypeError("gEval requires exactly one of criteria or evaluationSteps.");
536
+ }
537
+ if (options.criteria !== void 0 && options.criteria.trim().length === 0) {
538
+ throw new TypeError("gEval criteria must not be empty.");
539
+ }
540
+ if (options.evaluationSteps !== void 0 && options.evaluationSteps.length === 0) {
541
+ throw new TypeError("gEval evaluationSteps must not be empty.");
542
+ }
543
+ const config = metricConfig(options, options.name);
544
+ const rubric = validateRubric(options.rubric);
545
+ const scoreRange = rubric.length === 0 ? [0, 10] : [rubric[0]?.scoreRange[0] ?? 0, rubric.at(-1)?.scoreRange[1] ?? 10];
546
+ let generatedStepsPromise;
547
+ let generatedUsageClaimed = false;
548
+ async function resolveSteps() {
549
+ if (options.evaluationSteps !== void 0) {
550
+ return { steps: options.evaluationSteps, usage: Usage.empty() };
551
+ }
552
+ if (generatedStepsPromise === void 0) {
553
+ generatedStepsPromise = runJudge({
554
+ model: options.model,
555
+ schema: z.object({ steps: z.array(z.string()) }),
556
+ instructions: "Generate three or four concise evaluation steps from the criteria. Explain how the selected parameters should be judged in relation to one another.",
557
+ prompt: jsonPrompt({ criteria: options.criteria, parameters: options.evaluationParams }),
558
+ retries: config.retries
559
+ }).catch((error) => {
560
+ generatedStepsPromise = void 0;
561
+ throw error;
562
+ });
563
+ }
564
+ const result = await generatedStepsPromise;
565
+ if (result.data.steps.length === 0) throw new Error("G-Eval generated no evaluation steps.");
566
+ const usage = generatedUsageClaimed ? Usage.empty() : result.usage;
567
+ generatedUsageClaimed = true;
568
+ return { steps: result.data.steps, usage };
569
+ }
570
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
571
+ try {
572
+ const parameters = await resolveGEvalParameters(options, args);
573
+ const stepsResult = await resolveSteps();
574
+ const scoreResult = await runJudge({
575
+ model: options.model,
576
+ schema: z.object({ score: z.number(), reason: z.string() }),
577
+ instructions: config.strictMode ? "Apply the evaluation steps and return score 1 only for complete compliance, otherwise 0. Give a concise evidence-based reason." : `Apply the evaluation steps and return an integer score from ${scoreRange[0]} through ${scoreRange[1]}, plus a concise evidence-based reason.`,
578
+ prompt: jsonPrompt({
579
+ evaluationSteps: stepsResult.steps,
580
+ rubric,
581
+ parameters
582
+ }),
583
+ retries: config.retries
584
+ });
585
+ const rawScore = scoreResult.data.score;
586
+ if (!Number.isFinite(rawScore) || config.strictMode && rawScore !== 0 && rawScore !== 1 || !config.strictMode && (rawScore < scoreRange[0] || rawScore > scoreRange[1])) {
587
+ throw new RangeError(`G-Eval score ${rawScore} is outside the requested range.`);
588
+ }
589
+ const score = config.strictMode ? rawScore : (rawScore - scoreRange[0]) / (scoreRange[1] - scoreRange[0]);
590
+ const usage = addUsage(stepsResult.usage, scoreResult.usage);
591
+ return higherOutcome({
592
+ score,
593
+ threshold: config.threshold,
594
+ strictMode: config.strictMode,
595
+ comment: config.includeReason ? scoreResult.data.reason : void 0,
596
+ details: {
597
+ evaluationSteps: stepsResult.steps,
598
+ evaluationParams: options.evaluationParams,
599
+ rawScore,
600
+ scoreRange: [scoreRange[0], scoreRange[1]],
601
+ rubric: rubric.map((entry) => ({
602
+ scoreRange: [entry.scoreRange[0], entry.scoreRange[1]],
603
+ expectedOutcome: entry.expectedOutcome
604
+ }))
605
+ },
606
+ usage
607
+ });
608
+ } catch (error) {
609
+ return EvalOutcome.fromError(error);
610
+ }
611
+ });
612
+ }
613
+ function turnRelevancy(options) {
614
+ const config = metricConfig(options, "turn_relevancy");
615
+ const windowSize = validatePositiveInteger(options.windowSize ?? 10, "windowSize");
616
+ const concurrency = validatePositiveInteger(options.concurrency ?? 4, "concurrency");
617
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
618
+ try {
619
+ const turns = await resolveTurns(options.turns, args);
620
+ const interactions = unitInteractions(turns);
621
+ const windows = interactions.map(
622
+ (_, index) => interactions.slice(Math.max(0, index - windowSize + 1), index + 1).flat()
623
+ );
624
+ const verdictResults = await mapWithConcurrency(
625
+ windows,
626
+ concurrency,
627
+ (window) => runJudge({
628
+ model: options.model,
629
+ schema: binaryVerdictSchema,
630
+ instructions: "Judge whether the final assistant reply is relevant to the preceding conversation. Return yes for relevant and no for irrelevant, with a concise reason.",
631
+ prompt: jsonPrompt({ turns: window }),
632
+ retries: config.retries
633
+ })
634
+ );
635
+ const verdicts = verdictResults.map((result, index) => ({
636
+ interaction: index + 1,
637
+ ...result.data
638
+ }));
639
+ const score = verdicts.length === 0 ? 1 : verdicts.filter((verdict) => verdict.verdict === "yes").length / verdicts.length;
640
+ const reasonResult = await maybeReason({
641
+ model: options.model,
642
+ includeReason: config.includeReason,
643
+ retries: config.retries,
644
+ metric: "turn relevancy",
645
+ score,
646
+ evidence: { verdicts }
647
+ });
648
+ const usage = addUsage(...verdictResults.map((result) => result.usage), reasonResult.usage);
649
+ return higherOutcome({
650
+ score,
651
+ threshold: config.threshold,
652
+ strictMode: config.strictMode,
653
+ comment: reasonResult.reason,
654
+ details: { windowSize, concurrency, interactionCount: interactions.length, verdicts },
655
+ usage
656
+ });
657
+ } catch (error) {
658
+ return EvalOutcome.fromError(error);
659
+ }
660
+ });
661
+ }
662
+ function knowledgeRetention(options) {
663
+ const config = metricConfig(options, "knowledge_retention");
664
+ const concurrency = validatePositiveInteger(options.concurrency ?? 4, "concurrency");
665
+ return numericMetric(config.name, config, "higher_is_better", async (args) => {
666
+ try {
667
+ const turns = await resolveTurns(options.turns, args);
668
+ const userTurns = turns.map((turn, index) => ({ turn, index })).filter((entry) => entry.turn.role === "user");
669
+ const knowledgeResults = await mapWithConcurrency(
670
+ userTurns,
671
+ concurrency,
672
+ (entry) => runJudge({
673
+ model: options.model,
674
+ schema: factsSchema,
675
+ instructions: "Extract durable factual information newly supplied by the final user message. Use prior turns only to resolve references. Return concise facts; return an empty array when nothing new was supplied.",
676
+ prompt: jsonPrompt({
677
+ previousTurns: turns.slice(0, entry.index),
678
+ userMessage: entry.turn.content
679
+ }),
680
+ retries: config.retries
681
+ })
682
+ );
683
+ const knowledge = userTurns.map((entry, index) => ({
684
+ turnIndex: entry.index,
685
+ facts: knowledgeResults[index]?.data.facts ?? []
686
+ }));
687
+ const assistantChecks = turns.map((turn, index) => ({ turn, index })).filter((entry) => entry.turn.role === "assistant").map((entry) => ({
688
+ ...entry,
689
+ facts: knowledge.filter((item) => item.turnIndex < entry.index).flatMap((item) => item.facts)
690
+ })).filter((entry) => entry.facts.length > 0);
691
+ const verdictResults = await mapWithConcurrency(
692
+ assistantChecks,
693
+ concurrency,
694
+ (entry) => runJudge({
695
+ model: options.model,
696
+ schema: z.object({ attrition: z.boolean(), reason: z.string() }),
697
+ instructions: "Determine whether the assistant reply forgets, contradicts, or unnecessarily asks again for information already supplied by the user. Set attrition true only when knowledge was lost.",
698
+ prompt: jsonPrompt({ knownFacts: entry.facts, assistantReply: entry.turn.content }),
699
+ retries: config.retries
700
+ })
701
+ );
702
+ const verdicts = assistantChecks.map((entry, index) => ({
703
+ turnIndex: entry.index,
704
+ attrition: verdictResults[index]?.data.attrition ?? true,
705
+ reason: verdictResults[index]?.data.reason ?? "Missing knowledge-retention verdict."
706
+ }));
707
+ const score = verdicts.length === 0 ? 1 : verdicts.filter((verdict) => !verdict.attrition).length / verdicts.length;
708
+ const reasonResult = await maybeReason({
709
+ model: options.model,
710
+ includeReason: config.includeReason,
711
+ retries: config.retries,
712
+ metric: "knowledge retention",
713
+ score,
714
+ evidence: { verdicts }
715
+ });
716
+ const usage = addUsage(
717
+ ...knowledgeResults.map((result) => result.usage),
718
+ ...verdictResults.map((result) => result.usage),
719
+ reasonResult.usage
720
+ );
721
+ return higherOutcome({
722
+ score,
723
+ threshold: config.threshold,
724
+ strictMode: config.strictMode,
725
+ comment: reasonResult.reason,
726
+ details: { concurrency, knowledge, verdicts },
727
+ usage
728
+ });
729
+ } catch (error) {
730
+ return EvalOutcome.fromError(error);
731
+ }
732
+ });
733
+ }
734
+ function numericMetric(name, config, direction, evaluate) {
735
+ return {
736
+ name,
737
+ required: config.required,
738
+ direction,
739
+ threshold: config.strictMode === true ? direction === "higher_is_better" ? 1 : 0 : config.threshold,
740
+ dataType: "NUMERIC",
741
+ evaluate
742
+ };
743
+ }
744
+ function metricConfig(options, defaultName) {
745
+ return {
746
+ name: options.name ?? defaultName,
747
+ threshold: validateThreshold(options.threshold ?? 0.5),
748
+ strictMode: options.strictMode ?? false,
749
+ includeReason: options.includeReason ?? true,
750
+ retries: validateRetries(options.retries ?? 0),
751
+ required: options.required ?? true
752
+ };
753
+ }
754
+ function higherOutcome(args) {
755
+ const score = args.strictMode ? args.score === 1 ? 1 : 0 : args.score;
756
+ const threshold = args.strictMode ? 1 : args.threshold;
757
+ const options = {
758
+ comment: args.comment,
759
+ usage: args.usage,
760
+ metadata: evaluationMetadata(
761
+ {
762
+ ...args.details,
763
+ scoreDirection: "higher_is_better",
764
+ threshold,
765
+ strictMode: args.strictMode
766
+ },
767
+ args.usage
768
+ )
769
+ };
770
+ return score >= threshold ? EvalOutcome.pass(score, options) : EvalOutcome.fail(score, options);
771
+ }
772
+ function lowerOutcome(args) {
773
+ const score = args.strictMode ? args.score === 0 ? 0 : 1 : args.score;
774
+ const threshold = args.strictMode ? 0 : args.threshold;
775
+ const options = {
776
+ comment: args.comment,
777
+ usage: args.usage,
778
+ metadata: evaluationMetadata(
779
+ {
780
+ ...args.details,
781
+ scoreDirection: "lower_is_better",
782
+ threshold,
783
+ strictMode: args.strictMode
784
+ },
785
+ args.usage
786
+ )
787
+ };
788
+ return score <= threshold ? EvalOutcome.pass(score, options) : EvalOutcome.fail(score, options);
789
+ }
790
+ async function maybeReason(args) {
791
+ if (!args.includeReason) return { usage: Usage.empty() };
792
+ const result = await runJudge({
793
+ model: args.model,
794
+ schema: reasonSchema,
795
+ instructions: `Write a concise final explanation for the ${args.metric} score. Ground it only in the supplied evidence and do not repeat the numeric score.`,
796
+ prompt: jsonPrompt({ score: args.score, evidence: args.evidence }),
797
+ retries: args.retries
798
+ });
799
+ return { reason: result.data.reason, usage: result.usage };
800
+ }
801
+ async function resolveInput(selector, args) {
802
+ return selector === void 0 ? formatValue(args.case.input) : selector(args);
803
+ }
804
+ async function resolveStringList(selectorOrValue, fallback, args, label) {
805
+ const value = selectorOrValue === void 0 ? fallback : typeof selectorOrValue === "function" ? await selectorOrValue(args) : selectorOrValue;
806
+ if (!Array.isArray(value) || value.length === 0 || value.some((item) => typeof item !== "string")) {
807
+ throw new TypeError(`${label} must be a non-empty array of strings.`);
808
+ }
809
+ return value;
810
+ }
811
+ async function resolveGEvalParameters(options, args) {
812
+ const parameters = {};
813
+ for (const parameter of options.evaluationParams) {
814
+ if (parameter === "input") parameters.input = await resolveInput(options.input, args);
815
+ if (parameter === "actualOutput") {
816
+ parameters.actualOutput = await resolveActualText(options.actual, args);
817
+ }
818
+ if (parameter === "expectedOutput") {
819
+ const expected = options.expected === void 0 ? args.case.expected : await options.expected(args);
820
+ if (expected === void 0) throw new Error("G-Eval expectedOutput is missing.");
821
+ parameters.expectedOutput = toJsonValue(expected);
822
+ }
823
+ if (parameter === "context") {
824
+ parameters.context = await resolveStringList(
825
+ options.context,
826
+ args.case.context,
827
+ args,
828
+ "context"
829
+ );
830
+ }
831
+ if (parameter === "retrievalContext") {
832
+ parameters.retrievalContext = await resolveStringList(
833
+ options.retrievalContext,
834
+ args.case.retrievalContext,
835
+ args,
836
+ "retrievalContext"
837
+ );
838
+ }
839
+ if (parameter === "metadata") parameters.metadata = args.case.metadata ?? {};
840
+ }
841
+ return parameters;
842
+ }
843
+ function validateRubric(rubric) {
844
+ if (rubric === void 0 || rubric.length === 0) return [];
845
+ const sorted = [...rubric].sort((left, right) => left.scoreRange[0] - right.scoreRange[0]);
846
+ for (const [index, entry] of sorted.entries()) {
847
+ const [start, end] = entry.scoreRange;
848
+ if (!Number.isInteger(start) || !Number.isInteger(end) || start < 0 || end > 10 || start > end) {
849
+ throw new RangeError(
850
+ "G-Eval rubric score ranges must be ordered integers from 0 through 10."
851
+ );
852
+ }
853
+ if (entry.expectedOutcome.trim().length === 0) {
854
+ throw new TypeError("G-Eval rubric expectedOutcome must not be empty.");
855
+ }
856
+ const next = sorted[index + 1];
857
+ if (next !== void 0 && end >= next.scoreRange[0]) {
858
+ throw new RangeError("G-Eval rubric score ranges must not overlap.");
859
+ }
860
+ }
861
+ const first = sorted[0];
862
+ const last = sorted.at(-1);
863
+ if (first !== void 0 && last !== void 0 && first.scoreRange[0] === last.scoreRange[1]) {
864
+ throw new RangeError("G-Eval rubric score range must span more than one value.");
865
+ }
866
+ return sorted;
867
+ }
868
+ function truthsExtractionInstructions(limit) {
869
+ return limit === void 0 ? "Extract concise, factual, undisputed truths from the supplied source material, ordered by importance. Return them in the facts array." : `Extract at most ${limit} concise, factual, undisputed truths from the supplied source material, ordered by importance. Return them in the facts array.`;
870
+ }
871
+ function limitValues(values, limit) {
872
+ return limit === void 0 ? values : values.slice(0, limit);
873
+ }
874
+ function validateThreshold(value) {
875
+ if (!Number.isFinite(value) || value < 0 || value > 1) {
876
+ throw new RangeError("Eval metric threshold must be between 0 and 1.");
877
+ }
878
+ return value;
879
+ }
880
+ function validateRetries(value) {
881
+ if (!Number.isInteger(value) || value < 0) {
882
+ throw new RangeError("Eval metric retries must be a non-negative integer.");
883
+ }
884
+ return value;
885
+ }
886
+ function validatePositiveInteger(value, label) {
887
+ if (!Number.isInteger(value) || value < 1) {
888
+ throw new RangeError(`${label} must be a positive integer.`);
889
+ }
890
+ return value;
891
+ }
892
+ function validateOptionalNonNegativeInteger(value, label) {
893
+ if (value === void 0) return void 0;
894
+ if (!Number.isInteger(value) || value < 0) {
895
+ throw new RangeError(`${label} must be a non-negative integer.`);
896
+ }
897
+ return value;
898
+ }
899
+ function assertSameLength(label, inputs, outputs) {
900
+ if (inputs.length !== outputs.length) {
901
+ throw new Error(`${label} count ${outputs.length} did not match input count ${inputs.length}.`);
902
+ }
903
+ }
904
+ function serializeVerdicts(verdicts) {
905
+ return verdicts.map((verdict) => {
906
+ const serialized = { verdict: verdict.verdict };
907
+ if (verdict.reason !== void 0) serialized.reason = verdict.reason;
908
+ return serialized;
909
+ });
910
+ }
911
+ function jsonPrompt(value) {
912
+ return JSON.stringify(value, null, 2);
913
+ }
914
+ function toJsonValue(value) {
915
+ if (!isJsonValue(value)) {
916
+ throw new TypeError("G-Eval expectedOutput must be a JSON value.");
917
+ }
918
+ return cloneJsonValue(value);
919
+ }
920
+ function cloneJsonValue(value) {
921
+ if (value === null || typeof value !== "object") return value;
922
+ if (Array.isArray(value)) return value.map(cloneJsonValue);
923
+ return Object.fromEntries(
924
+ Object.entries(value).map(([key, item]) => {
925
+ if (item === void 0) {
926
+ throw new TypeError("G-Eval expectedOutput must be a JSON value.");
927
+ }
928
+ return [key, cloneJsonValue(item)];
929
+ })
930
+ );
931
+ }
932
+ function normalizeEvalTurns(value) {
933
+ const source = conversationArray(value);
934
+ if (source === void 0) return void 0;
935
+ const turns = [];
936
+ for (const entry of source) {
937
+ if (typeof entry !== "object" || entry === null) continue;
938
+ const role = entry.role;
939
+ if (role !== "user" && role !== "assistant") continue;
940
+ const content = entry.content;
941
+ const text = contentText(content);
942
+ if (text.length === 0) continue;
943
+ const metadata = entry.metadata;
944
+ turns.push(
945
+ typeof metadata === "object" && metadata !== null && !Array.isArray(metadata) ? { role, content: text, metadata } : { role, content: text }
946
+ );
947
+ }
948
+ return turns.length === 0 ? void 0 : turns;
949
+ }
950
+ async function resolveTurns(selector, args) {
951
+ const source = selector === void 0 ? args.output : await selector(args);
952
+ const turns = normalizeEvalTurns(source);
953
+ if (turns === void 0) {
954
+ throw new TypeError(
955
+ "Conversational eval requires non-empty EvalTurn[], Message[], or an output with messages."
956
+ );
957
+ }
958
+ return turns;
959
+ }
960
+ function unitInteractions(turns) {
961
+ const interactions = [];
962
+ let current = [];
963
+ let hasUser = false;
964
+ for (const turn of turns) {
965
+ if (current.at(-1)?.role === "assistant" && turn.role === "user" && hasUser) {
966
+ interactions.push(current);
967
+ current = [turn];
968
+ hasUser = true;
969
+ continue;
970
+ }
971
+ current.push(turn);
972
+ if (turn.role === "user") hasUser = true;
973
+ }
974
+ if (current.length > 1 && current.at(-1)?.role === "assistant" && hasUser) {
975
+ interactions.push(current);
976
+ }
977
+ return interactions;
978
+ }
979
+ function conversationArray(value) {
980
+ if (Array.isArray(value)) return value;
981
+ if (typeof value === "object" && value !== null && "messages" in value) {
982
+ const messages = value.messages;
983
+ return Array.isArray(messages) ? messages : void 0;
984
+ }
985
+ return void 0;
986
+ }
987
+ function contentText(content) {
988
+ if (typeof content === "string") return content;
989
+ if (!Array.isArray(content)) return "";
990
+ return content.flatMap(
991
+ (part) => typeof part === "object" && part !== null && "type" in part && part.type === "text" && "text" in part && typeof part.text === "string" ? [part.text] : []
992
+ ).join("\n");
993
+ }
994
+
995
+ export {
996
+ answerRelevancy,
997
+ promptAlignment,
998
+ jsonCorrectness,
999
+ hallucination,
1000
+ faithfulness,
1001
+ abstention,
1002
+ summarization,
1003
+ gEval,
1004
+ turnRelevancy,
1005
+ knowledgeRetention
1006
+ };
1007
+ //# sourceMappingURL=chunk-BUPC72Y2.js.map