@tangle-network/agent-eval 0.172.0 → 0.173.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (219) hide show
  1. package/CHANGELOG.md +49 -0
  2. package/README.md +17 -2
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/analyst/index.d.ts +13 -13
  5. package/dist/analyst/index.js +7 -6
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{attestation-CJBGmMVh.d.ts → attestation-c1QvaBdX.d.ts} +2 -2
  8. package/dist/{attestation-CJBGmMVh.d.ts.map → attestation-c1QvaBdX.d.ts.map} +1 -1
  9. package/dist/{backend-integrity-e79K3UPD.d.ts → backend-integrity-CeuTgqsd.d.ts} +3 -4
  10. package/dist/backend-integrity-CeuTgqsd.d.ts.map +1 -0
  11. package/dist/{benchmark-h-h4bfqj.d.ts → benchmark-BjLGkfnN.d.ts} +3 -3
  12. package/dist/{benchmark-h-h4bfqj.d.ts.map → benchmark-BjLGkfnN.d.ts.map} +1 -1
  13. package/dist/{benchmark-command-DoFcisuM.js → benchmark-command-9S20PRel.js} +9 -9
  14. package/dist/{benchmark-command-DoFcisuM.js.map → benchmark-command-9S20PRel.js.map} +1 -1
  15. package/dist/benchmarks/index.d.ts +5 -5
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/{bounded-process-CVOC_D3H.js → bounded-process-BBZob7vl.js} +64 -10
  18. package/dist/bounded-process-BBZob7vl.js.map +1 -0
  19. package/dist/builder-eval/index.d.ts +3 -3
  20. package/dist/builder-eval/index.js +2 -2
  21. package/dist/campaign/index.d.ts +9 -9
  22. package/dist/campaign/index.js +6 -6
  23. package/dist/{campaign-Dp35pBbS.js → campaign-pxS0wmo4.js} +8 -8
  24. package/dist/{campaign-Dp35pBbS.js.map → campaign-pxS0wmo4.js.map} +1 -1
  25. package/dist/chat-client-Db4bqYfA.js +115 -0
  26. package/dist/chat-client-Db4bqYfA.js.map +1 -0
  27. package/dist/{chat-json-call-5Jxna-aV.js → chat-json-call-C26igCih.js} +16 -6
  28. package/dist/chat-json-call-C26igCih.js.map +1 -0
  29. package/dist/cli.js +31 -17
  30. package/dist/cli.js.map +1 -1
  31. package/dist/{client-BvwNkIRN.js → client-BlLY6o2w.js} +2 -2
  32. package/dist/{client-BvwNkIRN.js.map → client-BlLY6o2w.js.map} +1 -1
  33. package/dist/{client-Df7wdslk.d.ts → client-DlqdbM7n.d.ts} +4 -4
  34. package/dist/{client-Df7wdslk.d.ts.map → client-DlqdbM7n.d.ts.map} +1 -1
  35. package/dist/contract/index.d.ts +13 -13
  36. package/dist/contract/index.js +202 -10
  37. package/dist/contract/index.js.map +1 -1
  38. package/dist/{counterfactual-Bee5_BIn.d.ts → counterfactual-CLgrwhkY.d.ts} +4 -4
  39. package/dist/{counterfactual-Bee5_BIn.d.ts.map → counterfactual-CLgrwhkY.d.ts.map} +1 -1
  40. package/dist/{chat-client-DI79OPye.js → default-registry-B0bKikCb.js} +3 -39
  41. package/dist/default-registry-B0bKikCb.js.map +1 -0
  42. package/dist/{default-registry-XxedTLwu.d.ts → default-registry-BKwc8bN5.d.ts} +6 -6
  43. package/dist/{default-registry-XxedTLwu.d.ts.map → default-registry-BKwc8bN5.d.ts.map} +1 -1
  44. package/dist/{define-agent-eval-0wW7gFhr.d.ts → define-agent-eval-CY6qdlGV.d.ts} +6 -6
  45. package/dist/{define-agent-eval-0wW7gFhr.d.ts.map → define-agent-eval-CY6qdlGV.d.ts.map} +1 -1
  46. package/dist/{define-agent-eval-jS8xj_Q_.js → define-agent-eval-D_i_s69h.js} +7 -7
  47. package/dist/{define-agent-eval-jS8xj_Q_.js.map → define-agent-eval-D_i_s69h.js.map} +1 -1
  48. package/dist/{dspy-rlm-engine-CS3qcCEk.js → dspy-rlm-engine-D5byiHn9.js} +5 -6
  49. package/dist/dspy-rlm-engine-D5byiHn9.js.map +1 -0
  50. package/dist/{emitter-Bvnu0VzL.d.ts → emitter-Cs0egaFd.d.ts} +3 -3
  51. package/dist/{emitter-Bvnu0VzL.d.ts.map → emitter-Cs0egaFd.d.ts.map} +1 -1
  52. package/dist/{engine-BfRay1qD.d.ts → engine-DhFir3Ys.d.ts} +23 -8
  53. package/dist/{engine-BfRay1qD.d.ts.map → engine-DhFir3Ys.d.ts.map} +1 -1
  54. package/dist/{eval-campaign-JDTeE6Pl.js → eval-campaign-BeAjdhzC.js} +2 -2
  55. package/dist/{eval-campaign-JDTeE6Pl.js.map → eval-campaign-BeAjdhzC.js.map} +1 -1
  56. package/dist/{exact-types-BEecmnWm.d.ts → exact-types-BKOEILRP.d.ts} +2 -2
  57. package/dist/{exact-types-BEecmnWm.d.ts.map → exact-types-BKOEILRP.d.ts.map} +1 -1
  58. package/dist/experiment/index.d.ts +4 -4
  59. package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts → external-optimizer-contracts-CQCpyrIL.d.ts} +2 -2
  60. package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts.map → external-optimizer-contracts-CQCpyrIL.d.ts.map} +1 -1
  61. package/dist/{external-optimizer-process-CQxylYeG.js → external-optimizer-process-Cq_Pg15r.js} +2 -2
  62. package/dist/{external-optimizer-process-CQxylYeG.js.map → external-optimizer-process-Cq_Pg15r.js.map} +1 -1
  63. package/dist/{external-optimizer-subprocess-Cex8Da2i.js → external-optimizer-subprocess-DgNebftP.js} +2 -2
  64. package/dist/{external-optimizer-subprocess-Cex8Da2i.js.map → external-optimizer-subprocess-DgNebftP.js.map} +1 -1
  65. package/dist/failure-cluster-OldNRoAt.d.ts +154 -0
  66. package/dist/failure-cluster-OldNRoAt.d.ts.map +1 -0
  67. package/dist/{feedback-trajectory-DIqpCyF0.d.ts → feedback-trajectory-CMnv_uYs.d.ts} +6 -6
  68. package/dist/{feedback-trajectory-DIqpCyF0.d.ts.map → feedback-trajectory-CMnv_uYs.d.ts.map} +1 -1
  69. package/dist/{heldout-gate-JgNRDZwZ.d.ts → heldout-gate-Df5hsqmm.d.ts} +7 -7
  70. package/dist/{heldout-gate-JgNRDZwZ.d.ts.map → heldout-gate-Df5hsqmm.d.ts.map} +1 -1
  71. package/dist/hosted/index.d.ts +2 -2
  72. package/dist/hosted/index.js +1 -1
  73. package/dist/{index-DnglhM0A.d.ts → index-BQqOjerE.d.ts} +94 -13
  74. package/dist/index-BQqOjerE.d.ts.map +1 -0
  75. package/dist/{index-_vPrVMRX.d.ts → index-CFDffsKz.d.ts} +11 -11
  76. package/dist/{index-_vPrVMRX.d.ts.map → index-CFDffsKz.d.ts.map} +1 -1
  77. package/dist/{index-DDAPhUJJ.d.ts → index-D0Db5X-4.d.ts} +45 -9
  78. package/dist/index-D0Db5X-4.d.ts.map +1 -0
  79. package/dist/{index-DMoxLG8P.d.ts → index-e7LXeRVa.d.ts} +3 -3
  80. package/dist/{index-DMoxLG8P.d.ts.map → index-e7LXeRVa.d.ts.map} +1 -1
  81. package/dist/index.d.ts +172 -46
  82. package/dist/index.d.ts.map +1 -1
  83. package/dist/index.js +64 -41
  84. package/dist/index.js.map +1 -1
  85. package/dist/{insight-report-08F022xN.d.ts → insight-report-DETqPc_A.d.ts} +4 -4
  86. package/dist/{insight-report-08F022xN.d.ts.map → insight-report-DETqPc_A.d.ts.map} +1 -1
  87. package/dist/{integrity-B_EDELom.d.ts → integrity-BKTcA-HP.d.ts} +3 -3
  88. package/dist/{integrity-B_EDELom.d.ts.map → integrity-BKTcA-HP.d.ts.map} +1 -1
  89. package/dist/{kind-factory-DMeEoMQZ.js → kind-factory-gP6lDySe.js} +164 -149
  90. package/dist/kind-factory-gP6lDySe.js.map +1 -0
  91. package/dist/{llm-client-BFMRpmqb.js → llm-client-CxQtdtd6.js} +12 -5
  92. package/dist/llm-client-CxQtdtd6.js.map +1 -0
  93. package/dist/{llm-judge-aQHIk5_-.js → llm-judge-B2YxbAJb.js} +81 -7
  94. package/dist/{llm-judge-aQHIk5_-.js.map → llm-judge-B2YxbAJb.js.map} +1 -1
  95. package/dist/{matrix-Ch8JO1pG.d.ts → matrix-DGu8KhSs.d.ts} +2 -2
  96. package/dist/{matrix-Ch8JO1pG.d.ts.map → matrix-DGu8KhSs.d.ts.map} +1 -1
  97. package/dist/meta-eval/index.d.ts +100 -5
  98. package/dist/meta-eval/index.d.ts.map +1 -1
  99. package/dist/meta-eval/index.js +200 -2
  100. package/dist/meta-eval/index.js.map +1 -1
  101. package/dist/{mint-DjfDUMHr.js → mint-vWOdD8Ae.js} +2 -2
  102. package/dist/{mint-DjfDUMHr.js.map → mint-vWOdD8Ae.js.map} +1 -1
  103. package/dist/multishot/golden/index.d.ts +1 -1
  104. package/dist/multishot/index.d.ts +2 -2
  105. package/dist/openapi.json +1 -1
  106. package/dist/pipelines/index.d.ts +5 -5
  107. package/dist/pipelines/index.js +3 -3
  108. package/dist/{pre-registration-DHz6P_6f.d.ts → pre-registration-BoI4ucR3.d.ts} +2 -2
  109. package/dist/{pre-registration-DHz6P_6f.d.ts.map → pre-registration-BoI4ucR3.d.ts.map} +1 -1
  110. package/dist/{produced-state-CxmbFxFd.js → produced-state-7VYDwtkk.js} +3 -3
  111. package/dist/{produced-state-CxmbFxFd.js.map → produced-state-7VYDwtkk.js.map} +1 -1
  112. package/dist/{promotion-policy-BBBcz5_3.d.ts → promotion-policy-CvMda3kU.d.ts} +2 -2
  113. package/dist/{promotion-policy-BBBcz5_3.d.ts.map → promotion-policy-CvMda3kU.d.ts.map} +1 -1
  114. package/dist/{provenance-Dp-vvyrU.d.ts → provenance-CRY67X50.d.ts} +39 -165
  115. package/dist/provenance-CRY67X50.d.ts.map +1 -0
  116. package/dist/{query-BPGMVlbM.js → query-D1nLIKt7.js} +2 -2
  117. package/dist/{query-BPGMVlbM.js.map → query-D1nLIKt7.js.map} +1 -1
  118. package/dist/{query-Na5gEIGd.d.ts → query-D6W6MaGx.d.ts} +3 -3
  119. package/dist/{query-Na5gEIGd.d.ts.map → query-D6W6MaGx.d.ts.map} +1 -1
  120. package/dist/{registry-xEb_xfns.d.ts → registry-7pOUBrtX.d.ts} +4 -4
  121. package/dist/{registry-xEb_xfns.d.ts.map → registry-7pOUBrtX.d.ts.map} +1 -1
  122. package/dist/{release-confidence-D6lQw_o7.d.ts → release-confidence-BAcNYOf1.d.ts} +4 -4
  123. package/dist/{release-confidence-D6lQw_o7.d.ts.map → release-confidence-BAcNYOf1.d.ts.map} +1 -1
  124. package/dist/{release-confidence-CzUHc4z4.js → release-confidence-BsGEg_xg.js} +3 -3
  125. package/dist/{release-confidence-CzUHc4z4.js.map → release-confidence-BsGEg_xg.js.map} +1 -1
  126. package/dist/reporting.d.ts +3 -3
  127. package/dist/reporting.js +1 -1
  128. package/dist/{researcher-CMUTQXD7.d.ts → researcher-jsW1X94L.d.ts} +7 -8
  129. package/dist/researcher-jsW1X94L.d.ts.map +1 -0
  130. package/dist/{reward-hacking-SkxYgT0x.js → reward-hacking-CKW4teig.js} +2 -2
  131. package/dist/{reward-hacking-SkxYgT0x.js.map → reward-hacking-CKW4teig.js.map} +1 -1
  132. package/dist/{reward-hacking-CgPRUesA.d.ts → reward-hacking-ZXEi9VCq.d.ts} +2 -2
  133. package/dist/{reward-hacking-CgPRUesA.d.ts.map → reward-hacking-ZXEi9VCq.d.ts.map} +1 -1
  134. package/dist/rl.d.ts +7 -7
  135. package/dist/rl.js +4 -4
  136. package/dist/rollout/index.d.ts +1 -1
  137. package/dist/rollout/index.js +2 -2
  138. package/dist/{rollout-Crypdx8s.js → rollout-C-znbbYg.js} +2 -2
  139. package/dist/{rollout-Crypdx8s.js.map → rollout-C-znbbYg.js.map} +1 -1
  140. package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts → rubric-predictive-validity-Dl1dvKCv.d.ts} +2 -2
  141. package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts.map → rubric-predictive-validity-Dl1dvKCv.d.ts.map} +1 -1
  142. package/dist/{run-record-DQjRcYwA.d.ts → run-record-DTv1MdjK.d.ts} +2 -2
  143. package/dist/{run-record-DQjRcYwA.d.ts.map → run-record-DTv1MdjK.d.ts.map} +1 -1
  144. package/dist/{run-record-DLORoL7t.js → run-record-ZIsR9Fif.js} +2 -2
  145. package/dist/{run-record-DLORoL7t.js.map → run-record-ZIsR9Fif.js.map} +1 -1
  146. package/dist/{schema-DID1Cqct.d.ts → schema-CR5cpjQ3.d.ts} +57 -2
  147. package/dist/{schema-DID1Cqct.d.ts.map → schema-CR5cpjQ3.d.ts.map} +1 -1
  148. package/dist/{schema-CdIX2aHu.js → schema-CSf6qWgZ.js} +35 -1
  149. package/dist/{schema-CdIX2aHu.js.map → schema-CSf6qWgZ.js.map} +1 -1
  150. package/dist/semantic-concept-judge-Ct3QU7t5.js +780 -0
  151. package/dist/semantic-concept-judge-Ct3QU7t5.js.map +1 -0
  152. package/dist/{series-convergence-D9WgpXGi.d.ts → series-convergence-DeG33RpC.d.ts} +2 -2
  153. package/dist/{series-convergence-D9WgpXGi.d.ts.map → series-convergence-DeG33RpC.d.ts.map} +1 -1
  154. package/dist/{server-CCEnywOR.js → server-BR6onwZB.js} +2 -2
  155. package/dist/{server-CCEnywOR.js.map → server-BR6onwZB.js.map} +1 -1
  156. package/dist/{skillopt-optimization-method-LHi02MzH.js → skillopt-optimization-method-BzdphODy.js} +6 -6
  157. package/dist/{skillopt-optimization-method-LHi02MzH.js.map → skillopt-optimization-method-BzdphODy.js.map} +1 -1
  158. package/dist/{statistical-heldout-Yldkntvy.d.ts → statistical-heldout-DTyB_6-1.d.ts} +3 -3
  159. package/dist/{statistical-heldout-Yldkntvy.d.ts.map → statistical-heldout-DTyB_6-1.d.ts.map} +1 -1
  160. package/dist/{store-Cq9oOrI1.d.ts → store-BErPvYBr.d.ts} +2 -2
  161. package/dist/{store-Cq9oOrI1.d.ts.map → store-BErPvYBr.d.ts.map} +1 -1
  162. package/dist/{store-otlp-CHjBvWQY.js → store-otlp-Dow0pk_5.js} +2 -2
  163. package/dist/{store-otlp-CHjBvWQY.js.map → store-otlp-Dow0pk_5.js.map} +1 -1
  164. package/dist/{store-tool-spans-BvdUbeOB.d.ts → store-tool-spans-CCZNsihA.d.ts} +8 -8
  165. package/dist/{store-tool-spans-BvdUbeOB.d.ts.map → store-tool-spans-CCZNsihA.d.ts.map} +1 -1
  166. package/dist/{store-tool-spans-B9o6tU8f.js → store-tool-spans-CeNj_m2L.js} +3 -3
  167. package/dist/{store-tool-spans-B9o6tU8f.js.map → store-tool-spans-CeNj_m2L.js.map} +1 -1
  168. package/dist/storyboard/index.d.ts +1 -1
  169. package/dist/{summary-report-DRstQNBX.d.ts → summary-report-gMrbYawB.d.ts} +3 -3
  170. package/dist/{summary-report-DRstQNBX.d.ts.map → summary-report-gMrbYawB.d.ts.map} +1 -1
  171. package/dist/{task-failure-attributes-CBGtLS_H.js → task-failure-attributes-CZjZeBsY.js} +3 -3
  172. package/dist/{task-failure-attributes-CBGtLS_H.js.map → task-failure-attributes-CZjZeBsY.js.map} +1 -1
  173. package/dist/{tool-groups-DjwlMBvW.d.ts → tool-groups-Cp4Xdzrp.d.ts} +3 -3
  174. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +1 -0
  175. package/dist/{tool-waste-CwGHzBzX.js → tool-waste-B9tdWV6g.js} +253 -6
  176. package/dist/tool-waste-B9tdWV6g.js.map +1 -0
  177. package/dist/{tool-waste-BrmLKxMw.d.ts → tool-waste-D23I0zWm.d.ts} +4 -4
  178. package/dist/{tool-waste-BrmLKxMw.d.ts.map → tool-waste-D23I0zWm.d.ts.map} +1 -1
  179. package/dist/trace-repair/index.d.ts +2 -2
  180. package/dist/traces.d.ts +10 -10
  181. package/dist/traces.js +7 -7
  182. package/dist/{trajectory-r1bQqvBQ.d.ts → trajectory-D7qrNvaN.d.ts} +3 -3
  183. package/dist/{trajectory-r1bQqvBQ.d.ts.map → trajectory-D7qrNvaN.d.ts.map} +1 -1
  184. package/dist/trajectory-replay/index.d.ts +3 -3
  185. package/dist/{types-CCZ34qmV.d.ts → types-BDV4PiMR.d.ts} +3 -3
  186. package/dist/{types-CCZ34qmV.d.ts.map → types-BDV4PiMR.d.ts.map} +1 -1
  187. package/dist/{types-nokrtr7M.d.ts → types-Ba5UQyVD.d.ts} +4 -4
  188. package/dist/{types-nokrtr7M.d.ts.map → types-Ba5UQyVD.d.ts.map} +1 -1
  189. package/dist/{types-DMoNFDWi.d.ts → types-DN2WdT5S.d.ts} +3 -3
  190. package/dist/{types-DMoNFDWi.d.ts.map → types-DN2WdT5S.d.ts.map} +1 -1
  191. package/dist/types-gvRsyJLh.d.ts +831 -0
  192. package/dist/types-gvRsyJLh.d.ts.map +1 -0
  193. package/dist/wire/index.d.ts +3 -3
  194. package/dist/wire/index.js +1 -1
  195. package/docs/plants.md +69 -1
  196. package/docs/public-api.md +6 -4
  197. package/docs/trace-analysis.md +55 -8
  198. package/package.json +1 -1
  199. package/dist/backend-integrity-e79K3UPD.d.ts.map +0 -1
  200. package/dist/bounded-process-CVOC_D3H.js.map +0 -1
  201. package/dist/chat-client-DI79OPye.js.map +0 -1
  202. package/dist/chat-json-call-5Jxna-aV.js.map +0 -1
  203. package/dist/dspy-rlm-engine-CS3qcCEk.js.map +0 -1
  204. package/dist/failure-cluster-6YSvsKlp.d.ts +0 -58
  205. package/dist/failure-cluster-6YSvsKlp.d.ts.map +0 -1
  206. package/dist/index-DDAPhUJJ.d.ts.map +0 -1
  207. package/dist/index-DnglhM0A.d.ts.map +0 -1
  208. package/dist/kind-factory-DMeEoMQZ.js.map +0 -1
  209. package/dist/llm-client-BFMRpmqb.js.map +0 -1
  210. package/dist/provenance-Dp-vvyrU.d.ts.map +0 -1
  211. package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
  212. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
  213. package/dist/researcher-CMUTQXD7.d.ts.map +0 -1
  214. package/dist/semantic-concept-judge-I36eejJx.js +0 -382
  215. package/dist/semantic-concept-judge-I36eejJx.js.map +0 -1
  216. package/dist/tool-groups-DjwlMBvW.d.ts.map +0 -1
  217. package/dist/tool-waste-CwGHzBzX.js.map +0 -1
  218. package/dist/types-Bfk0uxRj.d.ts +0 -443
  219. package/dist/types-Bfk0uxRj.d.ts.map +0 -1
@@ -0,0 +1,780 @@
1
+ import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
2
+ import { l as Mutex } from "./ledger-core-PIfjCbKn.js";
3
+ import { j as RawAnalystFindingSchema } from "./kind-factory-gP6lDySe.js";
4
+ import { c as extractJsonPayload } from "./llm-client-CxQtdtd6.js";
5
+ import { n as paidJsonChat, t as paidChat } from "./chat-json-call-C26igCih.js";
6
+ import { n as decodeRawFindingArray } from "./dspy-rlm-engine-D5byiHn9.js";
7
+ import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
8
+ import { z } from "zod";
9
+ import { dirname } from "node:path";
10
+ //#region src/analyst/chat-trace-engine.ts
11
+ /**
12
+ * A `TraceAnalysisEngine` that runs entirely inside Node against a
13
+ * caller-owned `ChatClient`.
14
+ *
15
+ * The other engine in this package, `createDspyRlmTraceEngine`, reaches the
16
+ * DSPy RLM through a Python subprocess. That made every model-backed analyst
17
+ * unreachable for a consumer that already owns a model seam and no Python:
18
+ * `createTraceAnalyst` requires an engine, the only exported constructor
19
+ * required a Python runner, so `buildDefaultAnalystRegistry()` registered the
20
+ * deterministic analyst alone and `analystsFromRegistry` refused the result.
21
+ *
22
+ * This engine closes that path. It drives the same investigation contract —
23
+ * bounded trace tools, a prose answer, a strict findings array — with native
24
+ * function calls over the transport the caller already bound with
25
+ * `createChatClient`. A caller holding a bare
26
+ * `(request: LlmCallRequest) => Promise<LlmCallResult>` adapts it in one line:
27
+ *
28
+ * createChatClient({ transport: 'custom', chat: call, defaultModel, maximumAttempts })
29
+ *
30
+ * agent-eval still executes no paid model and holds no provider credential.
31
+ * Every call goes through `paidChat`, so the shared cost ledger reserves the
32
+ * priced maximum before the call and settles the receipt after it.
33
+ *
34
+ * Shape of one investigation:
35
+ * 1. Investigation turns. The model reads the trace store through the tools
36
+ * its analyst kind was given. A turn that requests no tool ends the
37
+ * phase; an exhausted iteration or tool budget also ends it.
38
+ * 2. One report turn. The model returns `{ answer, findings }` as JSON,
39
+ * decoded by the same `decodeRawFindingArray` the Python bridge uses, so
40
+ * a row this engine accepts is a row that engine could report.
41
+ */
42
+ /** Bumped whenever this engine's execution behavior changes. */
43
+ const CHAT_TRACE_ENGINE_VERSION = "1.0.0";
44
+ const ENGINE_ID = "chat-trace";
45
+ /** Marker appended to a tool result cut down to the retained-output budget. */
46
+ const TRUNCATION_MARKER = "\n…[truncated to the analyst maxOutputChars budget]";
47
+ /**
48
+ * The report envelope. Its `findings` rows are generated from
49
+ * `RawAnalystFindingSchema`, so the schema offered to a provider and the
50
+ * decoder that accepts the answer cannot drift apart.
51
+ */
52
+ function buildReportJsonSchema() {
53
+ const row = z.toJSONSchema(RawAnalystFindingSchema, { target: "draft-7" });
54
+ delete row.$schema;
55
+ return {
56
+ name: "trace_analysis_report",
57
+ schema: {
58
+ type: "object",
59
+ additionalProperties: false,
60
+ required: ["answer", "findings"],
61
+ properties: {
62
+ answer: {
63
+ type: "string",
64
+ description: "Direct prose answer to the question."
65
+ },
66
+ findings: {
67
+ type: "array",
68
+ items: row
69
+ }
70
+ }
71
+ }
72
+ };
73
+ }
74
+ const REPORT_JSON_SCHEMA = buildReportJsonSchema();
75
+ /**
76
+ * Run bounded recursive trace analysis in-process over a caller-owned chat
77
+ * transport. No Python, no subprocess, no loopback proxy.
78
+ */
79
+ function createChatTraceEngine(options) {
80
+ const model = resolveModel(options);
81
+ const maxOutputTokens = options.maxOutputTokens ?? 16384;
82
+ if (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0) throw new TypeError("chat trace engine maxOutputTokens must be a positive safe integer");
83
+ if (options.temperature !== void 0 && (!Number.isFinite(options.temperature) || options.temperature < 0)) throw new TypeError("chat trace engine temperature must be a non-negative finite number");
84
+ if (options.requestTimeoutMs !== void 0 && (!Number.isSafeInteger(options.requestTimeoutMs) || options.requestTimeoutMs <= 0)) throw new TypeError("chat trace engine requestTimeoutMs must be a positive safe integer");
85
+ const reportFormat = options.reportFormat ?? "json-object";
86
+ if (reportFormat !== "json-object" && reportFormat !== "json-schema") throw new TypeError(`chat trace engine reportFormat must be json-object or json-schema`);
87
+ return {
88
+ id: ENGINE_ID,
89
+ description: "In-process recursive trace analysis over a caller-owned ChatClient with native tool calls.",
90
+ model,
91
+ version: CHAT_TRACE_ENGINE_VERSION,
92
+ executionConfig: {
93
+ kind: ENGINE_ID,
94
+ model,
95
+ transport: options.chat.transport,
96
+ maximum_attempts: options.chat.maximumAttempts ?? null,
97
+ pricing: options.pricing ? { ...options.pricing } : null,
98
+ max_output_tokens: maxOutputTokens,
99
+ temperature: options.temperature ?? null,
100
+ thinking: options.thinking ?? null,
101
+ request_timeout_ms: options.requestTimeoutMs ?? null,
102
+ report_format: reportFormat,
103
+ report_schema_name: REPORT_JSON_SCHEMA.name
104
+ },
105
+ analyze: (request) => analyze(request, {
106
+ ...options,
107
+ model,
108
+ maxOutputTokens,
109
+ reportFormat
110
+ })
111
+ };
112
+ }
113
+ async function analyze(request, options) {
114
+ if (request.limits.maxLlmCalls < 2) throw new Error(`chat trace engine reserves one model call for the report, so maxLlmCalls must be at least 2 (analyst '${request.analystId}' declared ${request.limits.maxLlmCalls})`);
115
+ const toolsByName = new Map(request.tools.map((tool) => [tool.name, tool]));
116
+ if (toolsByName.size !== request.tools.length) throw new Error(`chat trace engine received duplicate tool names for '${request.analystId}'`);
117
+ const toolDefinitions = request.tools.map((tool) => ({
118
+ type: "function",
119
+ function: {
120
+ name: tool.name,
121
+ description: tool.description,
122
+ parameters: tool.parameters
123
+ }
124
+ }));
125
+ const investigationTurns = Math.max(1, Math.min(request.limits.maxIterations, request.limits.maxLlmCalls - 1));
126
+ const messages = [{
127
+ role: "system",
128
+ content: systemPrompt(request, investigationTurns)
129
+ }];
130
+ if (request.taskInputs) messages.push({
131
+ role: "user",
132
+ content: renderTaskInputs(request.taskInputs)
133
+ });
134
+ messages.push({
135
+ role: "user",
136
+ content: request.question
137
+ });
138
+ request.log?.("trace analyst engine started", {
139
+ engine: ENGINE_ID,
140
+ model: options.model,
141
+ transport: options.chat.transport,
142
+ tools: request.tools.map((tool) => tool.name),
143
+ limits: request.limits
144
+ });
145
+ const trajectory = [];
146
+ const servedModels = /* @__PURE__ */ new Set();
147
+ let modelCalls = 0;
148
+ let toolCalls = 0;
149
+ let truncatedToolResults = 0;
150
+ let toolBudgetExhausted = false;
151
+ let turnsUsed = 0;
152
+ let stoppedOnAnswer = false;
153
+ for (let turn = 0; turn < investigationTurns; turn++) {
154
+ turnsUsed = turn + 1;
155
+ const response = await callModel(request, options, {
156
+ messages,
157
+ tools: toolDefinitions,
158
+ toolChoice: "auto",
159
+ purpose: "investigation"
160
+ });
161
+ modelCalls += 1;
162
+ servedModels.add(response.servedModel ?? "unreported");
163
+ const requested = response.toolCalls ?? [];
164
+ messages.push({
165
+ role: "assistant",
166
+ content: response.content,
167
+ ...requested.length > 0 ? { toolCalls: requested } : {}
168
+ });
169
+ trajectory.push({
170
+ turn: turnsUsed,
171
+ phase: "investigation",
172
+ content: response.content,
173
+ tool_calls: requested.map((call) => ({
174
+ id: call.id,
175
+ name: call.name
176
+ })),
177
+ finish_reason: response.finishReason ?? null
178
+ });
179
+ if (requested.length === 0) {
180
+ stoppedOnAnswer = true;
181
+ break;
182
+ }
183
+ for (const call of requested) {
184
+ if (toolCalls >= request.limits.maxToolCalls) {
185
+ toolBudgetExhausted = true;
186
+ messages.push(toolMessage(call, exhaustedToolBudget(request.limits.maxToolCalls)));
187
+ continue;
188
+ }
189
+ toolCalls += 1;
190
+ const executed = await executeTool(call, toolsByName, request);
191
+ if (executed.truncated) truncatedToolResults += 1;
192
+ messages.push(toolMessage(call, executed.payload));
193
+ trajectory.push({
194
+ turn: turnsUsed,
195
+ phase: "tool",
196
+ name: call.name,
197
+ ok: executed.ok,
198
+ truncated: executed.truncated,
199
+ result_chars: executed.payload.length
200
+ });
201
+ }
202
+ if (toolBudgetExhausted) break;
203
+ }
204
+ messages.push({
205
+ role: "user",
206
+ content: reportPrompt(options.reportFormat)
207
+ });
208
+ const report = await callModel(request, options, {
209
+ messages,
210
+ purpose: "report",
211
+ json: options.reportFormat
212
+ });
213
+ modelCalls += 1;
214
+ servedModels.add(report.servedModel ?? "unreported");
215
+ trajectory.push({
216
+ turn: turnsUsed + 1,
217
+ phase: "report",
218
+ content: report.content,
219
+ finish_reason: report.finishReason ?? null
220
+ });
221
+ const parsed = parseReport(report.content, request.analystId, (index, reason) => {
222
+ request.log?.("finding rejected: report row failed schema validation", {
223
+ engine: ENGINE_ID,
224
+ index,
225
+ reason
226
+ });
227
+ });
228
+ const result = {
229
+ answer: parsed.answer,
230
+ findings: parsed.findings,
231
+ trajectory,
232
+ modelCalls,
233
+ toolCalls,
234
+ runtime: {
235
+ engine: ENGINE_ID,
236
+ model: options.model,
237
+ transport: options.chat.transport,
238
+ report_format: options.reportFormat,
239
+ investigation_turns: turnsUsed,
240
+ investigation_stopped_on_answer: stoppedOnAnswer,
241
+ tool_budget_exhausted: toolBudgetExhausted,
242
+ truncated_tool_results: truncatedToolResults,
243
+ rejected_findings: parsed.rejectedFindings,
244
+ served_models: [...servedModels].sort(),
245
+ task_inputs: request.taskInputs ? "prompt-delivered" : "none"
246
+ }
247
+ };
248
+ request.log?.("trace analyst engine completed", {
249
+ engine: ENGINE_ID,
250
+ model_calls: result.modelCalls,
251
+ tool_calls: result.toolCalls,
252
+ findings: result.findings.length
253
+ });
254
+ return result;
255
+ }
256
+ async function callModel(request, options, turn) {
257
+ const chatRequest = {
258
+ model: options.model,
259
+ messages: turn.messages.map((message) => ({ ...message })),
260
+ maxTokens: options.maxOutputTokens,
261
+ ...turn.tools && turn.tools.length > 0 ? {
262
+ tools: turn.tools,
263
+ ...turn.toolChoice ? { toolChoice: turn.toolChoice } : {}
264
+ } : {},
265
+ ...turn.json ? { jsonMode: true } : {},
266
+ ...turn.json === "json-schema" ? { jsonSchema: REPORT_JSON_SCHEMA } : {},
267
+ ...options.temperature === void 0 ? {} : { temperature: options.temperature },
268
+ ...options.thinking === void 0 ? {} : { thinking: options.thinking },
269
+ ...options.requestTimeoutMs === void 0 ? {} : { timeoutMs: options.requestTimeoutMs }
270
+ };
271
+ const paid = await paidChat({
272
+ chat: options.chat,
273
+ request: chatRequest,
274
+ ledger: request.costLedger,
275
+ channel: "analyst",
276
+ phase: request.costPhase,
277
+ actor: request.analystId,
278
+ ...request.costTags ? { tags: request.costTags } : {},
279
+ ...options.pricing ? { pricing: options.pricing } : {},
280
+ ...request.signal ? { signal: request.signal } : {}
281
+ });
282
+ if (!paid.succeeded) throw new Error(`chat trace engine ${turn.purpose} call failed for '${request.analystId}': ${paid.error.message}`, { cause: paid.error });
283
+ return {
284
+ content: paid.response.content,
285
+ ...paid.response.toolCalls ? { toolCalls: paid.response.toolCalls } : {},
286
+ ...paid.response.servedModel === void 0 ? {} : { servedModel: paid.response.servedModel },
287
+ ...paid.response.finishReason === void 0 ? {} : { finishReason: paid.response.finishReason }
288
+ };
289
+ }
290
+ async function executeTool(call, toolsByName, request) {
291
+ const descriptor = toolsByName.get(call.name);
292
+ if (!descriptor) return {
293
+ ok: false,
294
+ truncated: false,
295
+ payload: toolError(`unknown tool '${call.name}'; available tools are ${[...toolsByName.keys()].sort().join(", ")}`)
296
+ };
297
+ let args;
298
+ try {
299
+ args = call.argumentsJson.trim() === "" ? {} : JSON.parse(call.argumentsJson);
300
+ } catch (error) {
301
+ return {
302
+ ok: false,
303
+ truncated: false,
304
+ payload: toolError(`arguments for '${call.name}' were not JSON: ${messageOf(error)}. Send valid JSON arguments.`)
305
+ };
306
+ }
307
+ try {
308
+ const value = await descriptor.handler(args, request.signal ? { signal: request.signal } : void 0);
309
+ return truncateToolPayload(JSON.stringify(value ?? null), request.limits.maxOutputChars);
310
+ } catch (error) {
311
+ if (request.signal?.aborted) throw error;
312
+ request.log?.("trace tool failed", {
313
+ engine: ENGINE_ID,
314
+ analyst_id: request.analystId,
315
+ tool: call.name,
316
+ reason: messageOf(error)
317
+ });
318
+ return {
319
+ ok: false,
320
+ truncated: false,
321
+ payload: toolError(messageOf(error))
322
+ };
323
+ }
324
+ }
325
+ function truncateToolPayload(payload, maxOutputChars) {
326
+ if (payload.length <= maxOutputChars) return {
327
+ ok: true,
328
+ truncated: false,
329
+ payload
330
+ };
331
+ const keep = Math.max(0, maxOutputChars - 50);
332
+ return {
333
+ ok: true,
334
+ truncated: true,
335
+ payload: `${payload.slice(0, keep)}${TRUNCATION_MARKER}`
336
+ };
337
+ }
338
+ function toolMessage(call, content) {
339
+ return {
340
+ role: "tool",
341
+ toolCallId: call.id,
342
+ content
343
+ };
344
+ }
345
+ function toolError(reason) {
346
+ return JSON.stringify({ error: reason });
347
+ }
348
+ function exhaustedToolBudget(maxToolCalls) {
349
+ return toolError(`the trace-tool budget of ${maxToolCalls} calls is spent; answer from the evidence already read`);
350
+ }
351
+ function systemPrompt(request, investigationTurns) {
352
+ return [request.instructions.trim(), [
353
+ "HOW THIS INVESTIGATION RUNS:",
354
+ `- You have ${investigationTurns} investigation turns and at most ${request.limits.maxToolCalls} trace-tool calls.`,
355
+ "- Call the trace tools to read the store. Never state a trace fact you did not read.",
356
+ `- A tool result longer than ${request.limits.maxOutputChars} characters is truncated; narrow the query instead of asking again.`,
357
+ "- A tool result carrying an \"error\" field is feedback: fix the call or take another route.",
358
+ "- Answer with no tool call once you have the evidence. You are then asked for the final report."
359
+ ].join("\n")].join("\n\n");
360
+ }
361
+ function reportPrompt(reportFormat) {
362
+ return [
363
+ "Report now. Do not call any more tools.",
364
+ "Return one JSON object with exactly two fields:",
365
+ " \"answer\": a direct prose answer to the question.",
366
+ " \"findings\": an array of findings in the schema above. Emit [] when there is nothing to report.",
367
+ reportFormat === "json-object" ? "Return the object alone, with no surrounding prose and no code fence." : "Return the object in the response schema you were given."
368
+ ].join("\n");
369
+ }
370
+ function renderTaskInputs(taskInputs) {
371
+ return ["TASK INPUTS — structured material delivered with the question, not fetched through tools:", JSON.stringify(taskInputs, null, 2)].join("\n");
372
+ }
373
+ function parseReport(content, analystId, onRejectedFinding) {
374
+ const envelope = coerceReportEnvelope(content, analystId);
375
+ if (typeof envelope.answer !== "string" || !envelope.answer.trim()) throw new Error(`chat trace engine report for '${analystId}' carried no answer`);
376
+ const decoded = decodeRawFindingArray(envelope.findings);
377
+ if (decoded.topLevelError !== void 0) throw new Error(`chat trace engine report for '${analystId}' had a malformed findings array: ${decoded.topLevelError}`);
378
+ for (const rejection of decoded.rejected) onRejectedFinding(rejection.index, `${rejection.code}${rejection.path ? ` at ${rejection.path}` : ""}: ${rejection.message}`);
379
+ return {
380
+ answer: envelope.answer,
381
+ findings: decoded.accepted,
382
+ rejectedFindings: decoded.rejected.length
383
+ };
384
+ }
385
+ function coerceReportEnvelope(content, analystId) {
386
+ let value;
387
+ try {
388
+ value = JSON.parse(extractJsonPayload(content));
389
+ } catch (error) {
390
+ throw new Error(`chat trace engine report for '${analystId}' was not JSON: ${messageOf(error)}`);
391
+ }
392
+ if (typeof value !== "object" || value === null || Array.isArray(value)) throw new Error(`chat trace engine report for '${analystId}' was not a JSON object`);
393
+ return value;
394
+ }
395
+ function resolveModel(options) {
396
+ const model = options.model ?? options.chat.defaultModel;
397
+ if (typeof model !== "string" || !model.trim() || model !== model.trim()) throw new TypeError("chat trace engine needs a model: pass ChatTraceEngineOptions.model or bind defaultModel on the ChatClient");
398
+ return model;
399
+ }
400
+ function messageOf(error) {
401
+ return error instanceof Error ? error.message : String(error);
402
+ }
403
+ //#endregion
404
+ //#region src/locked-jsonl-appender.ts
405
+ /**
406
+ * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
407
+ * payloads. The reference-replay store does the same thing for typed
408
+ * `ReferenceReplayRun` rows; this is the generic version used by
409
+ * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants
410
+ * append-only durable telemetry without rolling its own lock.
411
+ *
412
+ * Locks are per absolute file path (process-local). Cross-process
413
+ * concurrency is NOT addressed — that's an fcntl/flock problem.
414
+ */
415
+ const mutexes = /* @__PURE__ */ new Map();
416
+ function getMutex(path) {
417
+ let m = mutexes.get(path);
418
+ if (!m) {
419
+ m = new Mutex();
420
+ mutexes.set(path, m);
421
+ }
422
+ return m;
423
+ }
424
+ var LockedJsonlAppender = class {
425
+ path;
426
+ mutex;
427
+ constructor(path) {
428
+ this.path = path;
429
+ this.mutex = getMutex(path);
430
+ if (!existsSync(dirname(path))) mkdirSync(dirname(path), { recursive: true });
431
+ }
432
+ async append(entry) {
433
+ const line = `${JSON.stringify(entry)}\n`;
434
+ await this.mutex.runExclusive(() => {
435
+ appendFileSync(this.path, line);
436
+ });
437
+ }
438
+ };
439
+ //#endregion
440
+ //#region src/analyst/findings-store.ts
441
+ /**
442
+ * FindingsStore — durable persistence for AnalystFinding rows + a diff
443
+ * helper so we can answer "what changed since the last run?" without
444
+ * recomputing analysts.
445
+ *
446
+ * On-disk shape is JSONL: one finding per line, append-only, locked via
447
+ * LockedJsonlAppender. Operators get crash-safety (no partial JSON),
448
+ * cheap reads (sequential parse), and trivial backup (rsync the file).
449
+ *
450
+ * Reads are non-locking: a reader sees a consistent snapshot of all
451
+ * fully-written lines and skips an incomplete trailing line if the
452
+ * writer is mid-append. Cross-process locking is intentionally out of
453
+ * scope (see locked-jsonl-appender.ts).
454
+ *
455
+ * The store is run-scoped: callers pass `runId` on append and on load,
456
+ * which keeps multi-run files cleanly partitioned. The `diffFindings`
457
+ * helper compares two run-id sets using stable `finding_id` semantics —
458
+ * the diff is the cross-run signal the regression dashboard renders.
459
+ */
460
+ var FindingsStore = class {
461
+ path;
462
+ appender;
463
+ constructor(path) {
464
+ this.path = path;
465
+ this.appender = new LockedJsonlAppender(path);
466
+ }
467
+ async append(runId, findings) {
468
+ for (const f of findings) {
469
+ const row = {
470
+ ...f,
471
+ run_id: runId
472
+ };
473
+ await this.appender.append(row);
474
+ }
475
+ }
476
+ /** Load every persisted finding. Discards malformed trailing lines silently. */
477
+ loadAll() {
478
+ if (!existsSync(this.path)) return [];
479
+ const raw = readFileSync(this.path, "utf8");
480
+ if (!raw) return [];
481
+ const out = [];
482
+ for (const line of raw.split("\n")) {
483
+ if (!line) continue;
484
+ try {
485
+ out.push(JSON.parse(line));
486
+ } catch {}
487
+ }
488
+ return out;
489
+ }
490
+ /** Filter to a single run. */
491
+ loadRun(runId) {
492
+ return this.loadAll().filter((r) => r.run_id === runId);
493
+ }
494
+ };
495
+ /**
496
+ * Default materiality test. Deliberately narrow so LLM-reword churn
497
+ * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
498
+ */
499
+ function defaultIsMaterial(a, b) {
500
+ if (a.severity !== b.severity) return true;
501
+ if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > .05) return true;
502
+ if (a.evidence_refs.length !== b.evidence_refs.length) return true;
503
+ return false;
504
+ }
505
+ /**
506
+ * Diff two findings sets by stable finding_id. Callers typically load
507
+ * the two run-id slices from the same store and pass them in.
508
+ */
509
+ function diffFindings(previous, current, policy = {}) {
510
+ const isMaterial = policy.isMaterial ?? defaultIsMaterial;
511
+ const prevById = new Map(previous.map((f) => [f.finding_id, f]));
512
+ const curById = new Map(current.map((f) => [f.finding_id, f]));
513
+ const appeared = [];
514
+ const disappeared = [];
515
+ const persisted = [];
516
+ const changed = [];
517
+ for (const [id, cur] of curById) {
518
+ const prev = prevById.get(id);
519
+ if (!prev) {
520
+ appeared.push(cur);
521
+ continue;
522
+ }
523
+ if (isMaterial(prev, cur)) changed.push({
524
+ previous: prev,
525
+ current: cur
526
+ });
527
+ else persisted.push(cur);
528
+ }
529
+ for (const [id, prev] of prevById) if (!curById.has(id)) disappeared.push(prev);
530
+ return {
531
+ appeared,
532
+ disappeared,
533
+ persisted,
534
+ changed
535
+ };
536
+ }
537
+ //#endregion
538
+ //#region src/semantic-concept-judge.ts
539
+ const DEFAULT_COMPLEXITY_WEIGHTS = {
540
+ render: 1,
541
+ integrate: 2,
542
+ compute: 2.5
543
+ };
544
+ const SEMANTIC_CONCEPT_JUDGE_VERSION = "semantic-concept-judge-v1-2026-04-24";
545
+ const DEFAULT_MAX_SOURCE = 45e3;
546
+ const DEFAULT_MAX_HTML = 3e4;
547
+ const DEFAULT_MAX_PER_FILE = 2e4;
548
+ const DEFAULT_TIMEOUT = 3e5;
549
+ const DEFAULT_MAX_TOKENS = 16e3;
550
+ const DEFAULT_MODEL = "claude-sonnet-4-6";
551
+ const SEMANTIC_SCHEMA = {
552
+ type: "object",
553
+ additionalProperties: false,
554
+ required: ["summary", "concepts"],
555
+ properties: {
556
+ summary: {
557
+ type: "string",
558
+ minLength: 20,
559
+ maxLength: 600
560
+ },
561
+ concepts: {
562
+ type: "array",
563
+ minItems: 1,
564
+ items: {
565
+ type: "object",
566
+ additionalProperties: false,
567
+ required: [
568
+ "concept",
569
+ "present",
570
+ "score",
571
+ "evidence",
572
+ "severity"
573
+ ],
574
+ properties: {
575
+ concept: {
576
+ type: "string",
577
+ minLength: 1,
578
+ maxLength: 120
579
+ },
580
+ present: { type: "boolean" },
581
+ score: {
582
+ type: "number",
583
+ minimum: 0,
584
+ maximum: 10
585
+ },
586
+ evidence: {
587
+ type: "string",
588
+ minLength: 5,
589
+ maxLength: 400
590
+ },
591
+ severity: {
592
+ type: "string",
593
+ enum: [
594
+ "critical",
595
+ "major",
596
+ "minor",
597
+ "info"
598
+ ]
599
+ }
600
+ }
601
+ }
602
+ }
603
+ }
604
+ };
605
+ function truncate(body, cap, label) {
606
+ if (body.length <= cap) return body;
607
+ return `${body.slice(0, cap)}\n… [truncated ${body.length - cap} chars of ${label}]`;
608
+ }
609
+ function buildPrompt(input, opts) {
610
+ const sourceBlob = input.sourceFiles.filter((f) => f.content.length <= opts.maxPerFileChars).map((f) => `--- FILE: ${f.path} ---\n${f.content}`).join("\n\n");
611
+ const html = input.servedHtml ?? "";
612
+ return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.
613
+
614
+ You MUST distinguish:
615
+ (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),
616
+ (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),
617
+ (c) ABSENT (concept nowhere).
618
+
619
+ A comment like "// TODO: add mint button" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.
620
+
621
+ USER REQUEST (what the agent was asked to build):
622
+ ${input.userRequest}
623
+
624
+ ${input.artifactLabel ? `ARTIFACT METADATA:\n name: ${input.artifactLabel}\n description: ${input.artifactDescription ?? ""}\n\n` : ""}EXPECTED CONCEPTS (each must be graded independently):
625
+ ${input.expectedConcepts.map((c, i) => ` ${i + 1}. "${c.name}"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(" | ")}]` : ""}`).join("\n")}
626
+
627
+ ${html ? `SERVED HTML (what the preview returns when hit):\n${truncate(html, opts.maxHtmlChars, "HTML")}\n\n` : ""}SOURCE FILES (the agent's workdir):
628
+ ${truncate(sourceBlob, opts.maxSourceChars, "source")}
629
+
630
+ For EACH concept, return:
631
+ - concept: the concept name as given (match exactly)
632
+ - present: boolean — does a working implementation exist?
633
+ - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent
634
+ - evidence: cite "<file>:<line>" or "served-html:<selector>" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.
635
+ - severity:
636
+ "info" when present: true AND score >= 7
637
+ "minor" when present: true AND 4 <= score < 7
638
+ "major" when present: false OR score < 4
639
+ "critical" when the concept is not only absent but a core user flow depends on it
640
+
641
+ Also produce a "summary" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.
642
+
643
+ BE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.
644
+
645
+ Return STRICT JSON. No prose outside the JSON.`;
646
+ }
647
+ /**
648
+ * Run the semantic concept judge. Soft-fails to available=false on
649
+ * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat
650
+ * that as "skip" rather than "fail."
651
+ */
652
+ async function runSemanticConceptJudge(input, options) {
653
+ const start = Date.now();
654
+ const totalCount = input.expectedConcepts.length;
655
+ if (totalCount === 0) return {
656
+ kind: "semantic-concept",
657
+ version: SEMANTIC_CONCEPT_JUDGE_VERSION,
658
+ score: 0,
659
+ presentCount: 0,
660
+ totalCount: 0,
661
+ findings: [],
662
+ summary: "no expected concepts declared",
663
+ durationMs: 0,
664
+ costUsd: null,
665
+ available: false,
666
+ error: "no expected concepts declared"
667
+ };
668
+ const opts = {
669
+ chat: options.chat,
670
+ model: options.model ?? options.chat.defaultModel ?? DEFAULT_MODEL,
671
+ timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
672
+ maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
673
+ maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
674
+ maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
675
+ maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
676
+ ...options.pricing ? { pricing: options.pricing } : {},
677
+ costLedger: options.costLedger ?? new CostLedger(),
678
+ costPhase: options.costPhase ?? "judge.semantic-concept",
679
+ costTags: options.costTags ?? {},
680
+ signal: options.signal ?? new AbortController().signal,
681
+ weightConcepts: options.weightConcepts ?? "mean",
682
+ complexityWeights: {
683
+ ...DEFAULT_COMPLEXITY_WEIGHTS,
684
+ ...options.complexityWeights ?? {}
685
+ }
686
+ };
687
+ const weightForConcept = (spec) => {
688
+ if (opts.weightConcepts === "mean") return 1;
689
+ if (spec.weight != null) return spec.weight;
690
+ if (opts.weightConcepts === "complexity") return opts.complexityWeights[spec.complexity ?? "render"] ?? 1;
691
+ return 1;
692
+ };
693
+ const weightByName = new Map(input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]));
694
+ let receipt;
695
+ try {
696
+ const request = {
697
+ model: opts.model,
698
+ messages: [{
699
+ role: "system",
700
+ content: "You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation."
701
+ }, {
702
+ role: "user",
703
+ content: buildPrompt(input, opts)
704
+ }],
705
+ jsonSchema: {
706
+ name: "semantic_concept_judge",
707
+ schema: SEMANTIC_SCHEMA
708
+ },
709
+ temperature: 0,
710
+ maxTokens: opts.maxTokens,
711
+ timeoutMs: opts.timeoutMs
712
+ };
713
+ const paid = await paidJsonChat({
714
+ chat: opts.chat,
715
+ request,
716
+ ledger: opts.costLedger,
717
+ channel: "judge",
718
+ phase: opts.costPhase,
719
+ actor: "semantic-concept",
720
+ tags: opts.costTags,
721
+ signal: opts.signal,
722
+ ...opts.pricing ? { pricing: opts.pricing } : {}
723
+ });
724
+ receipt = paid.receipt;
725
+ if (!paid.succeeded) throw paid.error;
726
+ const { value } = paid;
727
+ if (!value?.concepts || !Array.isArray(value.concepts)) throw new Error("judge returned malformed response — expected array under \"concepts\"");
728
+ const findings = value.concepts.map((c) => ({
729
+ concept: String(c.concept),
730
+ present: Boolean(c.present),
731
+ score: Math.max(0, Math.min(10, Number(c.score ?? 0))),
732
+ evidence: String(c.evidence ?? ""),
733
+ severity: [
734
+ "critical",
735
+ "major",
736
+ "minor",
737
+ "info"
738
+ ].includes(c.severity) ? c.severity : "info"
739
+ }));
740
+ const presentCount = findings.filter((f) => f.present && f.score >= 7).length;
741
+ let weightSum = 0;
742
+ let weightedScoreSum = 0;
743
+ for (const f of findings) {
744
+ const w = weightByName.get(f.concept) ?? 1;
745
+ weightSum += w;
746
+ weightedScoreSum += w * f.score;
747
+ }
748
+ const scoreAvg = weightSum > 0 ? weightedScoreSum / weightSum : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length);
749
+ return {
750
+ kind: "semantic-concept",
751
+ version: SEMANTIC_CONCEPT_JUDGE_VERSION,
752
+ score: Number((scoreAvg / 10).toFixed(3)),
753
+ presentCount,
754
+ totalCount,
755
+ findings,
756
+ summary: String(value.summary ?? ""),
757
+ durationMs: Date.now() - start,
758
+ costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
759
+ available: true
760
+ };
761
+ } catch (err) {
762
+ return {
763
+ kind: "semantic-concept",
764
+ version: SEMANTIC_CONCEPT_JUDGE_VERSION,
765
+ score: 0,
766
+ presentCount: 0,
767
+ totalCount,
768
+ findings: [],
769
+ summary: "",
770
+ durationMs: Date.now() - start,
771
+ costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
772
+ available: false,
773
+ error: err instanceof Error ? err.message : String(err)
774
+ };
775
+ }
776
+ }
777
+ //#endregion
778
+ export { diffFindings as a, defaultIsMaterial as i, runSemanticConceptJudge as n, createChatTraceEngine as o, FindingsStore as r, SEMANTIC_CONCEPT_JUDGE_VERSION as t };
779
+
780
+ //# sourceMappingURL=semantic-concept-judge-Ct3QU7t5.js.map