@claude-flow/cli 3.32.2 → 3.32.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (655) hide show
  1. package/.claude/helpers/helpers.manifest.json +3 -3
  2. package/.claude/helpers/statusline.cjs +38 -38
  3. package/catalog-manifest.json +2 -2
  4. package/package.json +2 -2
  5. package/plugins/ruflo-metaharness/scripts/smoke.sh +11 -11
  6. package/dist/src/agenticow/speculative-exploration.d.ts +0 -148
  7. package/dist/src/agenticow/speculative-exploration.js +0 -218
  8. package/dist/src/appliance/gguf-engine.d.ts +0 -91
  9. package/dist/src/appliance/gguf-engine.js +0 -425
  10. package/dist/src/appliance/ruvllm-bridge.d.ts +0 -102
  11. package/dist/src/appliance/ruvllm-bridge.js +0 -292
  12. package/dist/src/appliance/rvfa-builder.d.ts +0 -44
  13. package/dist/src/appliance/rvfa-builder.js +0 -329
  14. package/dist/src/appliance/rvfa-distribution.d.ts +0 -97
  15. package/dist/src/appliance/rvfa-distribution.js +0 -370
  16. package/dist/src/appliance/rvfa-format.d.ts +0 -111
  17. package/dist/src/appliance/rvfa-format.js +0 -393
  18. package/dist/src/appliance/rvfa-runner.d.ts +0 -69
  19. package/dist/src/appliance/rvfa-runner.js +0 -237
  20. package/dist/src/appliance/rvfa-signing.d.ts +0 -123
  21. package/dist/src/appliance/rvfa-signing.js +0 -347
  22. package/dist/src/autopilot-state.d.ts +0 -77
  23. package/dist/src/autopilot-state.js +0 -271
  24. package/dist/src/benchmarks/gaia-agent-planning.smoke.d.ts +0 -18
  25. package/dist/src/benchmarks/gaia-agent-planning.smoke.js +0 -253
  26. package/dist/src/benchmarks/gaia-agent.d.ts +0 -198
  27. package/dist/src/benchmarks/gaia-agent.js +0 -651
  28. package/dist/src/benchmarks/gaia-causal-memory.d.ts +0 -133
  29. package/dist/src/benchmarks/gaia-causal-memory.js +0 -281
  30. package/dist/src/benchmarks/gaia-causal-memory.smoke.d.ts +0 -22
  31. package/dist/src/benchmarks/gaia-causal-memory.smoke.js +0 -300
  32. package/dist/src/benchmarks/gaia-convergence.d.ts +0 -138
  33. package/dist/src/benchmarks/gaia-convergence.js +0 -260
  34. package/dist/src/benchmarks/gaia-convergence.smoke.d.ts +0 -19
  35. package/dist/src/benchmarks/gaia-convergence.smoke.js +0 -246
  36. package/dist/src/benchmarks/gaia-critic.d.ts +0 -123
  37. package/dist/src/benchmarks/gaia-critic.js +0 -312
  38. package/dist/src/benchmarks/gaia-critic.smoke.d.ts +0 -21
  39. package/dist/src/benchmarks/gaia-critic.smoke.js +0 -327
  40. package/dist/src/benchmarks/gaia-decomposer.d.ts +0 -125
  41. package/dist/src/benchmarks/gaia-decomposer.js +0 -350
  42. package/dist/src/benchmarks/gaia-decomposer.smoke.d.ts +0 -21
  43. package/dist/src/benchmarks/gaia-decomposer.smoke.js +0 -228
  44. package/dist/src/benchmarks/gaia-e2e-smoke.d.ts +0 -27
  45. package/dist/src/benchmarks/gaia-e2e-smoke.js +0 -136
  46. package/dist/src/benchmarks/gaia-extract.smoke.d.ts +0 -45
  47. package/dist/src/benchmarks/gaia-extract.smoke.js +0 -242
  48. package/dist/src/benchmarks/gaia-hardness/features.d.ts +0 -46
  49. package/dist/src/benchmarks/gaia-hardness/features.js +0 -170
  50. package/dist/src/benchmarks/gaia-hardness/predictor.d.ts +0 -105
  51. package/dist/src/benchmarks/gaia-hardness/predictor.js +0 -260
  52. package/dist/src/benchmarks/gaia-hardness/predictor.smoke.d.ts +0 -20
  53. package/dist/src/benchmarks/gaia-hardness/predictor.smoke.js +0 -235
  54. package/dist/src/benchmarks/gaia-hardness/train-data-loader.d.ts +0 -51
  55. package/dist/src/benchmarks/gaia-hardness/train-data-loader.js +0 -179
  56. package/dist/src/benchmarks/gaia-judge.d.ts +0 -88
  57. package/dist/src/benchmarks/gaia-judge.js +0 -437
  58. package/dist/src/benchmarks/gaia-loader.d.ts +0 -87
  59. package/dist/src/benchmarks/gaia-loader.js +0 -326
  60. package/dist/src/benchmarks/gaia-tools/file_read.d.ts +0 -35
  61. package/dist/src/benchmarks/gaia-tools/file_read.js +0 -403
  62. package/dist/src/benchmarks/gaia-tools/grounded_query.d.ts +0 -126
  63. package/dist/src/benchmarks/gaia-tools/grounded_query.js +0 -225
  64. package/dist/src/benchmarks/gaia-tools/index.d.ts +0 -32
  65. package/dist/src/benchmarks/gaia-tools/index.js +0 -36
  66. package/dist/src/benchmarks/gaia-tools/types.d.ts +0 -62
  67. package/dist/src/benchmarks/gaia-tools/types.js +0 -12
  68. package/dist/src/benchmarks/gaia-tools/web_search.d.ts +0 -30
  69. package/dist/src/benchmarks/gaia-tools/web_search.js +0 -210
  70. package/dist/src/benchmarks/gaia-voting.d.ts +0 -88
  71. package/dist/src/benchmarks/gaia-voting.js +0 -297
  72. package/dist/src/benchmarks/gaia-voting.smoke.d.ts +0 -20
  73. package/dist/src/benchmarks/gaia-voting.smoke.js +0 -332
  74. package/dist/src/benchmarks/pretrain/index.d.ts +0 -58
  75. package/dist/src/benchmarks/pretrain/index.js +0 -404
  76. package/dist/src/business-pods/bbs-budget-tracker.d.ts +0 -139
  77. package/dist/src/business-pods/bbs-budget-tracker.js +0 -358
  78. package/dist/src/business-pods/domain-affinity-policy.d.ts +0 -47
  79. package/dist/src/business-pods/domain-affinity-policy.js +0 -65
  80. package/dist/src/business-pods/pod-schema.d.ts +0 -96
  81. package/dist/src/business-pods/pod-schema.js +0 -225
  82. package/dist/src/commands/advisor.d.ts +0 -15
  83. package/dist/src/commands/advisor.js +0 -94
  84. package/dist/src/commands/agent-wasm.d.ts +0 -14
  85. package/dist/src/commands/agent-wasm.js +0 -333
  86. package/dist/src/commands/agent.d.ts +0 -8
  87. package/dist/src/commands/agent.js +0 -927
  88. package/dist/src/commands/analyze.d.ts +0 -19
  89. package/dist/src/commands/analyze.js +0 -2047
  90. package/dist/src/commands/announcements.d.ts +0 -17
  91. package/dist/src/commands/announcements.js +0 -0
  92. package/dist/src/commands/appliance-advanced.d.ts +0 -9
  93. package/dist/src/commands/appliance-advanced.js +0 -215
  94. package/dist/src/commands/appliance.d.ts +0 -8
  95. package/dist/src/commands/appliance.js +0 -404
  96. package/dist/src/commands/autopilot.d.ts +0 -15
  97. package/dist/src/commands/autopilot.js +0 -407
  98. package/dist/src/commands/benchmark.d.ts +0 -10
  99. package/dist/src/commands/benchmark.js +0 -460
  100. package/dist/src/commands/claims.d.ts +0 -10
  101. package/dist/src/commands/claims.js +0 -620
  102. package/dist/src/commands/cleanup.d.ts +0 -13
  103. package/dist/src/commands/cleanup.js +0 -250
  104. package/dist/src/commands/completions.d.ts +0 -10
  105. package/dist/src/commands/completions.js +0 -539
  106. package/dist/src/commands/config.d.ts +0 -8
  107. package/dist/src/commands/config.js +0 -428
  108. package/dist/src/commands/daemon.d.ts +0 -59
  109. package/dist/src/commands/daemon.js +0 -1733
  110. package/dist/src/commands/deployment.d.ts +0 -10
  111. package/dist/src/commands/deployment.js +0 -672
  112. package/dist/src/commands/doctor.d.ts +0 -10
  113. package/dist/src/commands/doctor.js +0 -1695
  114. package/dist/src/commands/eject.d.ts +0 -33
  115. package/dist/src/commands/eject.js +0 -195
  116. package/dist/src/commands/embeddings.d.ts +0 -18
  117. package/dist/src/commands/embeddings.js +0 -1623
  118. package/dist/src/commands/funnel.d.ts +0 -18
  119. package/dist/src/commands/funnel.js +0 -302
  120. package/dist/src/commands/gaia-bench.d.ts +0 -40
  121. package/dist/src/commands/gaia-bench.js +0 -597
  122. package/dist/src/commands/guidance.d.ts +0 -8
  123. package/dist/src/commands/guidance.js +0 -556
  124. package/dist/src/commands/hive-mind.d.ts +0 -11
  125. package/dist/src/commands/hive-mind.js +0 -1319
  126. package/dist/src/commands/hooks.d.ts +0 -8
  127. package/dist/src/commands/hooks.js +0 -4606
  128. package/dist/src/commands/index.d.ts +0 -118
  129. package/dist/src/commands/index.js +0 -364
  130. package/dist/src/commands/init.d.ts +0 -13
  131. package/dist/src/commands/init.js +0 -1297
  132. package/dist/src/commands/issues.d.ts +0 -21
  133. package/dist/src/commands/issues.js +0 -567
  134. package/dist/src/commands/mcp.d.ts +0 -11
  135. package/dist/src/commands/mcp.js +0 -732
  136. package/dist/src/commands/memory-backup.d.ts +0 -11
  137. package/dist/src/commands/memory-backup.js +0 -46
  138. package/dist/src/commands/memory-distill.d.ts +0 -27
  139. package/dist/src/commands/memory-distill.js +0 -374
  140. package/dist/src/commands/memory.d.ts +0 -8
  141. package/dist/src/commands/memory.js +0 -1596
  142. package/dist/src/commands/metaharness.d.ts +0 -39
  143. package/dist/src/commands/metaharness.js +0 -215
  144. package/dist/src/commands/migrate.d.ts +0 -8
  145. package/dist/src/commands/migrate.js +0 -742
  146. package/dist/src/commands/neural.d.ts +0 -10
  147. package/dist/src/commands/neural.js +0 -4331
  148. package/dist/src/commands/performance.d.ts +0 -10
  149. package/dist/src/commands/performance.js +0 -583
  150. package/dist/src/commands/plugins.d.ts +0 -11
  151. package/dist/src/commands/plugins.js +0 -826
  152. package/dist/src/commands/process.d.ts +0 -10
  153. package/dist/src/commands/process.js +0 -694
  154. package/dist/src/commands/progress.d.ts +0 -11
  155. package/dist/src/commands/progress.js +0 -259
  156. package/dist/src/commands/providers.d.ts +0 -10
  157. package/dist/src/commands/providers.js +0 -502
  158. package/dist/src/commands/proxy.d.ts +0 -21
  159. package/dist/src/commands/proxy.js +0 -310
  160. package/dist/src/commands/route.d.ts +0 -16
  161. package/dist/src/commands/route.js +0 -822
  162. package/dist/src/commands/ruvector/backup.d.ts +0 -11
  163. package/dist/src/commands/ruvector/backup.js +0 -747
  164. package/dist/src/commands/ruvector/benchmark.d.ts +0 -11
  165. package/dist/src/commands/ruvector/benchmark.js +0 -490
  166. package/dist/src/commands/ruvector/import.d.ts +0 -18
  167. package/dist/src/commands/ruvector/import.js +0 -373
  168. package/dist/src/commands/ruvector/index.d.ts +0 -29
  169. package/dist/src/commands/ruvector/index.js +0 -129
  170. package/dist/src/commands/ruvector/init.d.ts +0 -11
  171. package/dist/src/commands/ruvector/init.js +0 -467
  172. package/dist/src/commands/ruvector/migrate.d.ts +0 -11
  173. package/dist/src/commands/ruvector/migrate.js +0 -498
  174. package/dist/src/commands/ruvector/optimize.d.ts +0 -11
  175. package/dist/src/commands/ruvector/optimize.js +0 -505
  176. package/dist/src/commands/ruvector/pg-utils.d.ts +0 -14
  177. package/dist/src/commands/ruvector/pg-utils.js +0 -41
  178. package/dist/src/commands/ruvector/setup.d.ts +0 -18
  179. package/dist/src/commands/ruvector/setup.js +0 -765
  180. package/dist/src/commands/ruvector/status.d.ts +0 -11
  181. package/dist/src/commands/ruvector/status.js +0 -479
  182. package/dist/src/commands/security.d.ts +0 -10
  183. package/dist/src/commands/security.js +0 -1012
  184. package/dist/src/commands/session.d.ts +0 -8
  185. package/dist/src/commands/session.js +0 -757
  186. package/dist/src/commands/settings.d.ts +0 -19
  187. package/dist/src/commands/settings.js +0 -180
  188. package/dist/src/commands/spinner.d.ts +0 -16
  189. package/dist/src/commands/spinner.js +0 -329
  190. package/dist/src/commands/start.d.ts +0 -8
  191. package/dist/src/commands/start.js +0 -418
  192. package/dist/src/commands/status.d.ts +0 -8
  193. package/dist/src/commands/status.js +0 -608
  194. package/dist/src/commands/swarm.d.ts +0 -8
  195. package/dist/src/commands/swarm.js +0 -891
  196. package/dist/src/commands/task.d.ts +0 -8
  197. package/dist/src/commands/task.js +0 -675
  198. package/dist/src/commands/transfer-store.d.ts +0 -13
  199. package/dist/src/commands/transfer-store.js +0 -428
  200. package/dist/src/commands/update.d.ts +0 -8
  201. package/dist/src/commands/update.js +0 -276
  202. package/dist/src/commands/verify.d.ts +0 -19
  203. package/dist/src/commands/verify.js +0 -261
  204. package/dist/src/commands/version.d.ts +0 -42
  205. package/dist/src/commands/version.js +0 -106
  206. package/dist/src/commands/workflow.d.ts +0 -8
  207. package/dist/src/commands/workflow.js +0 -617
  208. package/dist/src/config/harness-feedback-applier.d.ts +0 -50
  209. package/dist/src/config/harness-feedback-applier.js +0 -122
  210. package/dist/src/config/proven-config-refresh.d.ts +0 -39
  211. package/dist/src/config/proven-config-refresh.js +0 -154
  212. package/dist/src/config/proven-config-rvfa.d.ts +0 -23
  213. package/dist/src/config/proven-config-rvfa.js +0 -73
  214. package/dist/src/config/proven-config.d.ts +0 -86
  215. package/dist/src/config/proven-config.js +0 -176
  216. package/dist/src/config-adapter.d.ts +0 -15
  217. package/dist/src/config-adapter.js +0 -186
  218. package/dist/src/encryption/vault.d.ts +0 -94
  219. package/dist/src/encryption/vault.js +0 -172
  220. package/dist/src/fs-secure.d.ts +0 -86
  221. package/dist/src/fs-secure.js +0 -133
  222. package/dist/src/funnel/advisor-tip.d.ts +0 -58
  223. package/dist/src/funnel/advisor-tip.js +0 -92
  224. package/dist/src/funnel/attribution.d.ts +0 -37
  225. package/dist/src/funnel/attribution.js +0 -101
  226. package/dist/src/funnel/consent.d.ts +0 -22
  227. package/dist/src/funnel/consent.js +0 -58
  228. package/dist/src/funnel/credit-errors.d.ts +0 -31
  229. package/dist/src/funnel/credit-errors.js +0 -88
  230. package/dist/src/funnel/credit-notifier.d.ts +0 -44
  231. package/dist/src/funnel/credit-notifier.js +0 -74
  232. package/dist/src/funnel/disclosure.d.ts +0 -47
  233. package/dist/src/funnel/disclosure.js +0 -109
  234. package/dist/src/funnel/enrollment.d.ts +0 -36
  235. package/dist/src/funnel/enrollment.js +0 -64
  236. package/dist/src/funnel/environment.d.ts +0 -17
  237. package/dist/src/funnel/environment.js +0 -39
  238. package/dist/src/funnel/event-transport.d.ts +0 -51
  239. package/dist/src/funnel/event-transport.js +0 -199
  240. package/dist/src/funnel/events.d.ts +0 -42
  241. package/dist/src/funnel/events.js +0 -150
  242. package/dist/src/funnel/index.d.ts +0 -21
  243. package/dist/src/funnel/index.js +0 -21
  244. package/dist/src/funnel/insights.d.ts +0 -50
  245. package/dist/src/funnel/insights.js +0 -120
  246. package/dist/src/funnel/local-signals.d.ts +0 -20
  247. package/dist/src/funnel/local-signals.js +0 -77
  248. package/dist/src/funnel/message-transport.d.ts +0 -51
  249. package/dist/src/funnel/message-transport.js +0 -149
  250. package/dist/src/funnel/messages.d.ts +0 -55
  251. package/dist/src/funnel/messages.js +0 -160
  252. package/dist/src/funnel/payout.d.ts +0 -40
  253. package/dist/src/funnel/payout.js +0 -60
  254. package/dist/src/funnel/power-saver-notifier.d.ts +0 -44
  255. package/dist/src/funnel/power-saver-notifier.js +0 -92
  256. package/dist/src/funnel/precedence.d.ts +0 -16
  257. package/dist/src/funnel/precedence.js +0 -85
  258. package/dist/src/funnel/promo.d.ts +0 -41
  259. package/dist/src/funnel/promo.js +0 -144
  260. package/dist/src/funnel/rate-limit-notifier.d.ts +0 -55
  261. package/dist/src/funnel/rate-limit-notifier.js +0 -102
  262. package/dist/src/funnel/rotation.d.ts +0 -19
  263. package/dist/src/funnel/rotation.js +0 -70
  264. package/dist/src/funnel/state.d.ts +0 -13
  265. package/dist/src/funnel/state.js +0 -52
  266. package/dist/src/funnel/toggle-cooldown.d.ts +0 -17
  267. package/dist/src/funnel/toggle-cooldown.js +0 -32
  268. package/dist/src/funnel/types.d.ts +0 -98
  269. package/dist/src/funnel/types.js +0 -26
  270. package/dist/src/index.d.ts +0 -81
  271. package/dist/src/index.js +0 -609
  272. package/dist/src/infrastructure/in-memory-repositories.d.ts +0 -68
  273. package/dist/src/infrastructure/in-memory-repositories.js +0 -264
  274. package/dist/src/init/claudemd-generator.d.ts +0 -16
  275. package/dist/src/init/claudemd-generator.js +0 -368
  276. package/dist/src/init/executor.d.ts +0 -41
  277. package/dist/src/init/executor.js +0 -2142
  278. package/dist/src/init/helper-refresh.d.ts +0 -79
  279. package/dist/src/init/helper-refresh.js +0 -347
  280. package/dist/src/init/helper-signing.d.ts +0 -37
  281. package/dist/src/init/helper-signing.js +0 -67
  282. package/dist/src/init/helpers-generator.d.ts +0 -87
  283. package/dist/src/init/helpers-generator.js +0 -1456
  284. package/dist/src/init/index.d.ts +0 -13
  285. package/dist/src/init/index.js +0 -15
  286. package/dist/src/init/mcp-generator.d.ts +0 -26
  287. package/dist/src/init/mcp-generator.js +0 -126
  288. package/dist/src/init/memory-package-resolver.d.ts +0 -53
  289. package/dist/src/init/memory-package-resolver.js +0 -118
  290. package/dist/src/init/settings-generator.d.ts +0 -14
  291. package/dist/src/init/settings-generator.js +0 -462
  292. package/dist/src/init/statusline-generator.d.ts +0 -28
  293. package/dist/src/init/statusline-generator.js +0 -177
  294. package/dist/src/init/types.d.ts +0 -331
  295. package/dist/src/init/types.js +0 -276
  296. package/dist/src/log-filters.d.ts +0 -46
  297. package/dist/src/log-filters.js +0 -107
  298. package/dist/src/mcp-client.d.ts +0 -92
  299. package/dist/src/mcp-client.js +0 -397
  300. package/dist/src/mcp-server.d.ts +0 -163
  301. package/dist/src/mcp-server.js +0 -751
  302. package/dist/src/mcp-tools/agent-execute-core.d.ts +0 -115
  303. package/dist/src/mcp-tools/agent-execute-core.js +0 -587
  304. package/dist/src/mcp-tools/agent-tools.d.ts +0 -9
  305. package/dist/src/mcp-tools/agent-tools.js +0 -837
  306. package/dist/src/mcp-tools/agentbbs-tools.d.ts +0 -28
  307. package/dist/src/mcp-tools/agentbbs-tools.js +0 -394
  308. package/dist/src/mcp-tools/agentdb-tools.d.ts +0 -35
  309. package/dist/src/mcp-tools/agentdb-tools.js +0 -1473
  310. package/dist/src/mcp-tools/agenticow-loader.d.ts +0 -59
  311. package/dist/src/mcp-tools/agenticow-loader.js +0 -105
  312. package/dist/src/mcp-tools/agenticow-speculate-tools.d.ts +0 -24
  313. package/dist/src/mcp-tools/agenticow-speculate-tools.js +0 -209
  314. package/dist/src/mcp-tools/agenticow-tools.d.ts +0 -36
  315. package/dist/src/mcp-tools/agenticow-tools.js +0 -360
  316. package/dist/src/mcp-tools/analyze-tools.d.ts +0 -38
  317. package/dist/src/mcp-tools/analyze-tools.js +0 -346
  318. package/dist/src/mcp-tools/auto-install.d.ts +0 -83
  319. package/dist/src/mcp-tools/auto-install.js +0 -131
  320. package/dist/src/mcp-tools/autopilot-tools.d.ts +0 -12
  321. package/dist/src/mcp-tools/autopilot-tools.js +0 -231
  322. package/dist/src/mcp-tools/browser-intent-tools.d.ts +0 -162
  323. package/dist/src/mcp-tools/browser-intent-tools.js +0 -548
  324. package/dist/src/mcp-tools/browser-session-tools.d.ts +0 -27
  325. package/dist/src/mcp-tools/browser-session-tools.js +0 -398
  326. package/dist/src/mcp-tools/browser-tools.d.ts +0 -21
  327. package/dist/src/mcp-tools/browser-tools.js +0 -760
  328. package/dist/src/mcp-tools/business-pod-tools.d.ts +0 -20
  329. package/dist/src/mcp-tools/business-pod-tools.js +0 -169
  330. package/dist/src/mcp-tools/claims-tools.d.ts +0 -12
  331. package/dist/src/mcp-tools/claims-tools.js +0 -863
  332. package/dist/src/mcp-tools/config-tools.d.ts +0 -8
  333. package/dist/src/mcp-tools/config-tools.js +0 -411
  334. package/dist/src/mcp-tools/coordination-tools.d.ts +0 -13
  335. package/dist/src/mcp-tools/coordination-tools.js +0 -729
  336. package/dist/src/mcp-tools/daa-tools.d.ts +0 -13
  337. package/dist/src/mcp-tools/daa-tools.js +0 -534
  338. package/dist/src/mcp-tools/embeddings-tools.d.ts +0 -9
  339. package/dist/src/mcp-tools/embeddings-tools.js +0 -904
  340. package/dist/src/mcp-tools/github-tools.d.ts +0 -9
  341. package/dist/src/mcp-tools/github-tools.js +0 -659
  342. package/dist/src/mcp-tools/guidance-tools.d.ts +0 -15
  343. package/dist/src/mcp-tools/guidance-tools.js +0 -639
  344. package/dist/src/mcp-tools/hive-mind-tools.d.ts +0 -8
  345. package/dist/src/mcp-tools/hive-mind-tools.js +0 -953
  346. package/dist/src/mcp-tools/hooks-tools.d.ts +0 -65
  347. package/dist/src/mcp-tools/hooks-tools.js +0 -4836
  348. package/dist/src/mcp-tools/http-fetch-tools.d.ts +0 -55
  349. package/dist/src/mcp-tools/http-fetch-tools.js +0 -329
  350. package/dist/src/mcp-tools/index.d.ts +0 -34
  351. package/dist/src/mcp-tools/index.js +0 -40
  352. package/dist/src/mcp-tools/managed-agent-tools.d.ts +0 -22
  353. package/dist/src/mcp-tools/managed-agent-tools.js +0 -357
  354. package/dist/src/mcp-tools/memory-tools.d.ts +0 -14
  355. package/dist/src/mcp-tools/memory-tools.js +0 -1330
  356. package/dist/src/mcp-tools/metaharness-tools.d.ts +0 -51
  357. package/dist/src/mcp-tools/metaharness-tools.js +0 -684
  358. package/dist/src/mcp-tools/neural-tools.d.ts +0 -54
  359. package/dist/src/mcp-tools/neural-tools.js +0 -1168
  360. package/dist/src/mcp-tools/performance-tools.d.ts +0 -16
  361. package/dist/src/mcp-tools/performance-tools.js +0 -675
  362. package/dist/src/mcp-tools/progress-tools.d.ts +0 -14
  363. package/dist/src/mcp-tools/progress-tools.js +0 -348
  364. package/dist/src/mcp-tools/request-tracker.d.ts +0 -17
  365. package/dist/src/mcp-tools/request-tracker.js +0 -27
  366. package/dist/src/mcp-tools/ruvllm-tools.d.ts +0 -9
  367. package/dist/src/mcp-tools/ruvllm-tools.js +0 -355
  368. package/dist/src/mcp-tools/security-tools.d.ts +0 -18
  369. package/dist/src/mcp-tools/security-tools.js +0 -556
  370. package/dist/src/mcp-tools/session-tools.d.ts +0 -8
  371. package/dist/src/mcp-tools/session-tools.js +0 -517
  372. package/dist/src/mcp-tools/swarm-tools.d.ts +0 -37
  373. package/dist/src/mcp-tools/swarm-tools.js +0 -390
  374. package/dist/src/mcp-tools/system-tools.d.ts +0 -13
  375. package/dist/src/mcp-tools/system-tools.js +0 -688
  376. package/dist/src/mcp-tools/task-tools.d.ts +0 -8
  377. package/dist/src/mcp-tools/task-tools.js +0 -487
  378. package/dist/src/mcp-tools/terminal-tools.d.ts +0 -8
  379. package/dist/src/mcp-tools/terminal-tools.js +0 -306
  380. package/dist/src/mcp-tools/testgen-tools.d.ts +0 -26
  381. package/dist/src/mcp-tools/testgen-tools.js +0 -168
  382. package/dist/src/mcp-tools/tool-loop-guardrail.d.ts +0 -31
  383. package/dist/src/mcp-tools/tool-loop-guardrail.js +0 -71
  384. package/dist/src/mcp-tools/transfer-tools.d.ts +0 -14
  385. package/dist/src/mcp-tools/transfer-tools.js +0 -447
  386. package/dist/src/mcp-tools/types.d.ts +0 -8
  387. package/dist/src/mcp-tools/types.js +0 -8
  388. package/dist/src/mcp-tools/validate-input.d.ts +0 -9
  389. package/dist/src/mcp-tools/validate-input.js +0 -9
  390. package/dist/src/mcp-tools/wasm-agent-tools.d.ts +0 -13
  391. package/dist/src/mcp-tools/wasm-agent-tools.js +0 -840
  392. package/dist/src/mcp-tools/workflow-tools.d.ts +0 -8
  393. package/dist/src/mcp-tools/workflow-tools.js +0 -884
  394. package/dist/src/memory/bge-embedder.d.ts +0 -25
  395. package/dist/src/memory/bge-embedder.js +0 -121
  396. package/dist/src/memory/cross-encoder-rerank.d.ts +0 -33
  397. package/dist/src/memory/cross-encoder-rerank.js +0 -123
  398. package/dist/src/memory/embedding-policy.d.ts +0 -21
  399. package/dist/src/memory/embedding-policy.js +0 -30
  400. package/dist/src/memory/embedding-quantization.d.ts +0 -62
  401. package/dist/src/memory/embedding-quantization.js +0 -156
  402. package/dist/src/memory/ewc-consolidation.d.ts +0 -305
  403. package/dist/src/memory/ewc-consolidation.js +0 -611
  404. package/dist/src/memory/graph-edge-writer.d.ts +0 -95
  405. package/dist/src/memory/graph-edge-writer.js +0 -217
  406. package/dist/src/memory/hybrid-retrieval.d.ts +0 -77
  407. package/dist/src/memory/hybrid-retrieval.js +0 -192
  408. package/dist/src/memory/intelligence.d.ts +0 -405
  409. package/dist/src/memory/intelligence.js +0 -1316
  410. package/dist/src/memory/lucene-bm25.d.ts +0 -19
  411. package/dist/src/memory/lucene-bm25.js +0 -308
  412. package/dist/src/memory/memory-bridge.d.ts +0 -537
  413. package/dist/src/memory/memory-bridge.js +0 -2460
  414. package/dist/src/memory/memory-initializer.d.ts +0 -556
  415. package/dist/src/memory/memory-initializer.js +0 -3021
  416. package/dist/src/memory/neural-package-bridge.d.ts +0 -48
  417. package/dist/src/memory/neural-package-bridge.js +0 -87
  418. package/dist/src/memory/rabitq-index.d.ts +0 -60
  419. package/dist/src/memory/rabitq-index.js +0 -242
  420. package/dist/src/memory/sona-optimizer.d.ts +0 -267
  421. package/dist/src/memory/sona-optimizer.js +0 -779
  422. package/dist/src/memory/structured-distill.d.ts +0 -48
  423. package/dist/src/memory/structured-distill.js +0 -125
  424. package/dist/src/output.d.ts +0 -9
  425. package/dist/src/output.js +0 -9
  426. package/dist/src/parser.d.ts +0 -89
  427. package/dist/src/parser.js +0 -516
  428. package/dist/src/plugins/manager.d.ts +0 -133
  429. package/dist/src/plugins/manager.js +0 -415
  430. package/dist/src/plugins/store/discovery.d.ts +0 -99
  431. package/dist/src/plugins/store/discovery.js +0 -1224
  432. package/dist/src/plugins/store/index.d.ts +0 -76
  433. package/dist/src/plugins/store/index.js +0 -141
  434. package/dist/src/plugins/store/search.d.ts +0 -46
  435. package/dist/src/plugins/store/search.js +0 -230
  436. package/dist/src/plugins/store/types.d.ts +0 -279
  437. package/dist/src/plugins/store/types.js +0 -7
  438. package/dist/src/plugins/tests/demo-plugin-store.d.ts +0 -7
  439. package/dist/src/plugins/tests/demo-plugin-store.js +0 -126
  440. package/dist/src/plugins/tests/standalone-test.d.ts +0 -12
  441. package/dist/src/plugins/tests/standalone-test.js +0 -188
  442. package/dist/src/plugins/tests/test-plugin-store.d.ts +0 -7
  443. package/dist/src/plugins/tests/test-plugin-store.js +0 -206
  444. package/dist/src/production/circuit-breaker.d.ts +0 -101
  445. package/dist/src/production/circuit-breaker.js +0 -241
  446. package/dist/src/production/error-handler.d.ts +0 -92
  447. package/dist/src/production/error-handler.js +0 -299
  448. package/dist/src/production/index.d.ts +0 -23
  449. package/dist/src/production/index.js +0 -18
  450. package/dist/src/production/monitoring.d.ts +0 -161
  451. package/dist/src/production/monitoring.js +0 -356
  452. package/dist/src/production/rate-limiter.d.ts +0 -80
  453. package/dist/src/production/rate-limiter.js +0 -201
  454. package/dist/src/production/retry.d.ts +0 -48
  455. package/dist/src/production/retry.js +0 -179
  456. package/dist/src/prompt.d.ts +0 -44
  457. package/dist/src/prompt.js +0 -501
  458. package/dist/src/runtime/headless.d.ts +0 -60
  459. package/dist/src/runtime/headless.js +0 -284
  460. package/dist/src/runtime/parent-death-watchdog.d.ts +0 -42
  461. package/dist/src/runtime/parent-death-watchdog.js +0 -70
  462. package/dist/src/ruvector/agent-wasm.d.ts +0 -228
  463. package/dist/src/ruvector/agent-wasm.js +0 -463
  464. package/dist/src/ruvector/ast-analyzer.d.ts +0 -67
  465. package/dist/src/ruvector/ast-analyzer.js +0 -277
  466. package/dist/src/ruvector/codemods/engine.d.ts +0 -45
  467. package/dist/src/ruvector/codemods/engine.js +0 -291
  468. package/dist/src/ruvector/codemods/scope-analysis.d.ts +0 -29
  469. package/dist/src/ruvector/codemods/scope-analysis.js +0 -162
  470. package/dist/src/ruvector/coverage-router.d.ts +0 -160
  471. package/dist/src/ruvector/coverage-router.js +0 -531
  472. package/dist/src/ruvector/coverage-tools.d.ts +0 -33
  473. package/dist/src/ruvector/coverage-tools.js +0 -157
  474. package/dist/src/ruvector/diff-classifier.d.ts +0 -175
  475. package/dist/src/ruvector/diff-classifier.js +0 -699
  476. package/dist/src/ruvector/diskann-backend.d.ts +0 -78
  477. package/dist/src/ruvector/diskann-backend.js +0 -310
  478. package/dist/src/ruvector/enhanced-model-router.d.ts +0 -172
  479. package/dist/src/ruvector/enhanced-model-router.js +0 -577
  480. package/dist/src/ruvector/graph-analyzer.d.ts +0 -187
  481. package/dist/src/ruvector/graph-analyzer.js +0 -929
  482. package/dist/src/ruvector/graph-backend.d.ts +0 -79
  483. package/dist/src/ruvector/graph-backend.js +0 -220
  484. package/dist/src/ruvector/index.d.ts +0 -38
  485. package/dist/src/ruvector/index.js +0 -86
  486. package/dist/src/ruvector/lora-adapter.d.ts +0 -292
  487. package/dist/src/ruvector/lora-adapter.js +0 -710
  488. package/dist/src/ruvector/model-prices.d.ts +0 -50
  489. package/dist/src/ruvector/model-prices.js +0 -72
  490. package/dist/src/ruvector/model-router.d.ts +0 -408
  491. package/dist/src/ruvector/model-router.js +0 -1162
  492. package/dist/src/ruvector/neural-router.d.ts +0 -182
  493. package/dist/src/ruvector/neural-router.js +0 -825
  494. package/dist/src/ruvector/output-verifier.d.ts +0 -83
  495. package/dist/src/ruvector/output-verifier.js +0 -277
  496. package/dist/src/ruvector/q-learning-router.d.ts +0 -227
  497. package/dist/src/ruvector/q-learning-router.js +0 -721
  498. package/dist/src/ruvector/router-calibrator.d.ts +0 -65
  499. package/dist/src/ruvector/router-calibrator.js +0 -120
  500. package/dist/src/ruvector/router-parallel-recorder.d.ts +0 -127
  501. package/dist/src/ruvector/router-parallel-recorder.js +0 -183
  502. package/dist/src/ruvector/router-trajectory.d.ts +0 -194
  503. package/dist/src/ruvector/router-trajectory.js +0 -281
  504. package/dist/src/ruvector/run-transcript-recorder.d.ts +0 -154
  505. package/dist/src/ruvector/run-transcript-recorder.js +0 -209
  506. package/dist/src/ruvector/ruvllm-wasm.d.ts +0 -179
  507. package/dist/src/ruvector/ruvllm-wasm.js +0 -379
  508. package/dist/src/ruvector/semantic-router.d.ts +0 -77
  509. package/dist/src/ruvector/semantic-router.js +0 -178
  510. package/dist/src/ruvector/task-embedder.d.ts +0 -56
  511. package/dist/src/ruvector/task-embedder.js +0 -237
  512. package/dist/src/ruvector/trajectory-tree.d.ts +0 -113
  513. package/dist/src/ruvector/trajectory-tree.js +0 -237
  514. package/dist/src/ruvector/vector-db.d.ts +0 -73
  515. package/dist/src/ruvector/vector-db.js +0 -301
  516. package/dist/src/ruvector/wasm-embedder.d.ts +0 -13
  517. package/dist/src/ruvector/wasm-embedder.js +0 -143
  518. package/dist/src/security/builtin-aidefence.d.ts +0 -34
  519. package/dist/src/security/builtin-aidefence.js +0 -86
  520. package/dist/src/services/agentic-flow-bridge.d.ts +0 -50
  521. package/dist/src/services/agentic-flow-bridge.js +0 -95
  522. package/dist/src/services/ai-job-dedup.d.ts +0 -61
  523. package/dist/src/services/ai-job-dedup.js +0 -136
  524. package/dist/src/services/checkpoint-gate.d.ts +0 -140
  525. package/dist/src/services/checkpoint-gate.js +0 -223
  526. package/dist/src/services/claim-service.d.ts +0 -204
  527. package/dist/src/services/claim-service.js +0 -818
  528. package/dist/src/services/config-file-manager.d.ts +0 -37
  529. package/dist/src/services/config-file-manager.js +0 -224
  530. package/dist/src/services/container-worker-pool.d.ts +0 -204
  531. package/dist/src/services/container-worker-pool.js +0 -589
  532. package/dist/src/services/daemon-autostart.d.ts +0 -17
  533. package/dist/src/services/daemon-autostart.js +0 -102
  534. package/dist/src/services/distill-oracle.d.ts +0 -190
  535. package/dist/src/services/distill-oracle.js +0 -349
  536. package/dist/src/services/distill-tuning.d.ts +0 -111
  537. package/dist/src/services/distill-tuning.js +0 -510
  538. package/dist/src/services/evolve-proof.d.ts +0 -253
  539. package/dist/src/services/evolve-proof.js +0 -343
  540. package/dist/src/services/fable-harness.d.ts +0 -208
  541. package/dist/src/services/fable-harness.js +0 -388
  542. package/dist/src/services/git-workspace-identity.d.ts +0 -42
  543. package/dist/src/services/git-workspace-identity.js +0 -99
  544. package/dist/src/services/global-ai-budget.d.ts +0 -135
  545. package/dist/src/services/global-ai-budget.js +0 -415
  546. package/dist/src/services/harness-benchmark.d.ts +0 -68
  547. package/dist/src/services/harness-benchmark.js +0 -94
  548. package/dist/src/services/harness-canary.d.ts +0 -60
  549. package/dist/src/services/harness-canary.js +0 -69
  550. package/dist/src/services/harness-corpus-harvester.d.ts +0 -51
  551. package/dist/src/services/harness-corpus-harvester.js +0 -74
  552. package/dist/src/services/harness-flywheel-generations.d.ts +0 -115
  553. package/dist/src/services/harness-flywheel-generations.js +0 -360
  554. package/dist/src/services/harness-flywheel-runtime.d.ts +0 -25
  555. package/dist/src/services/harness-flywheel-runtime.js +0 -92
  556. package/dist/src/services/harness-flywheel.d.ts +0 -48
  557. package/dist/src/services/harness-flywheel.js +0 -207
  558. package/dist/src/services/harness-frozen-eval.d.ts +0 -22
  559. package/dist/src/services/harness-frozen-eval.js +0 -66
  560. package/dist/src/services/harness-hosts.d.ts +0 -40
  561. package/dist/src/services/harness-hosts.js +0 -88
  562. package/dist/src/services/harness-improvement-ledger.d.ts +0 -63
  563. package/dist/src/services/harness-improvement-ledger.js +0 -101
  564. package/dist/src/services/harness-loop.d.ts +0 -53
  565. package/dist/src/services/harness-loop.js +0 -85
  566. package/dist/src/services/harness-qualification.d.ts +0 -66
  567. package/dist/src/services/harness-qualification.js +0 -143
  568. package/dist/src/services/harness-replay.d.ts +0 -37
  569. package/dist/src/services/harness-replay.js +0 -92
  570. package/dist/src/services/harness-verify.d.ts +0 -33
  571. package/dist/src/services/harness-verify.js +0 -26
  572. package/dist/src/services/harness-worker.d.ts +0 -23
  573. package/dist/src/services/harness-worker.js +0 -66
  574. package/dist/src/services/headless-worker-executor.d.ts +0 -358
  575. package/dist/src/services/headless-worker-executor.js +0 -1269
  576. package/dist/src/services/index.d.ts +0 -13
  577. package/dist/src/services/index.js +0 -11
  578. package/dist/src/services/memory-backup.d.ts +0 -53
  579. package/dist/src/services/memory-backup.js +0 -198
  580. package/dist/src/services/memory-distillation.d.ts +0 -41
  581. package/dist/src/services/memory-distillation.js +0 -277
  582. package/dist/src/services/native-training.d.ts +0 -68
  583. package/dist/src/services/native-training.js +0 -141
  584. package/dist/src/services/registry-api.d.ts +0 -58
  585. package/dist/src/services/registry-api.js +0 -146
  586. package/dist/src/services/repo-supervisor.d.ts +0 -70
  587. package/dist/src/services/repo-supervisor.js +0 -228
  588. package/dist/src/services/ruvector-training.d.ts +0 -222
  589. package/dist/src/services/ruvector-training.js +0 -693
  590. package/dist/src/services/swarm-memory-branches.d.ts +0 -135
  591. package/dist/src/services/swarm-memory-branches.js +0 -213
  592. package/dist/src/services/weight-eft.d.ts +0 -305
  593. package/dist/src/services/weight-eft.js +0 -296
  594. package/dist/src/services/worker-daemon.d.ts +0 -439
  595. package/dist/src/services/worker-daemon.js +0 -1865
  596. package/dist/src/services/worker-queue.d.ts +0 -194
  597. package/dist/src/services/worker-queue.js +0 -513
  598. package/dist/src/services/workspace-lease.d.ts +0 -55
  599. package/dist/src/services/workspace-lease.js +0 -191
  600. package/dist/src/suggest.d.ts +0 -53
  601. package/dist/src/suggest.js +0 -200
  602. package/dist/src/transfer/anonymization/index.d.ts +0 -25
  603. package/dist/src/transfer/anonymization/index.js +0 -175
  604. package/dist/src/transfer/deploy-seraphine.d.ts +0 -13
  605. package/dist/src/transfer/deploy-seraphine.js +0 -205
  606. package/dist/src/transfer/export.d.ts +0 -25
  607. package/dist/src/transfer/export.js +0 -113
  608. package/dist/src/transfer/index.d.ts +0 -12
  609. package/dist/src/transfer/index.js +0 -31
  610. package/dist/src/transfer/ipfs/client.d.ts +0 -109
  611. package/dist/src/transfer/ipfs/client.js +0 -307
  612. package/dist/src/transfer/ipfs/upload.d.ts +0 -95
  613. package/dist/src/transfer/ipfs/upload.js +0 -413
  614. package/dist/src/transfer/models/seraphine.d.ts +0 -72
  615. package/dist/src/transfer/models/seraphine.js +0 -373
  616. package/dist/src/transfer/serialization/cfp.d.ts +0 -49
  617. package/dist/src/transfer/serialization/cfp.js +0 -183
  618. package/dist/src/transfer/storage/gcs.d.ts +0 -82
  619. package/dist/src/transfer/storage/gcs.js +0 -272
  620. package/dist/src/transfer/storage/index.d.ts +0 -6
  621. package/dist/src/transfer/storage/index.js +0 -6
  622. package/dist/src/transfer/store/discovery.d.ts +0 -84
  623. package/dist/src/transfer/store/discovery.js +0 -382
  624. package/dist/src/transfer/store/download.d.ts +0 -70
  625. package/dist/src/transfer/store/download.js +0 -334
  626. package/dist/src/transfer/store/index.d.ts +0 -84
  627. package/dist/src/transfer/store/index.js +0 -153
  628. package/dist/src/transfer/store/publish.d.ts +0 -76
  629. package/dist/src/transfer/store/publish.js +0 -294
  630. package/dist/src/transfer/store/registry.d.ts +0 -58
  631. package/dist/src/transfer/store/registry.js +0 -285
  632. package/dist/src/transfer/store/search.d.ts +0 -54
  633. package/dist/src/transfer/store/search.js +0 -232
  634. package/dist/src/transfer/store/tests/standalone-test.d.ts +0 -12
  635. package/dist/src/transfer/store/tests/standalone-test.js +0 -190
  636. package/dist/src/transfer/store/types.d.ts +0 -193
  637. package/dist/src/transfer/store/types.js +0 -6
  638. package/dist/src/transfer/test-seraphine.d.ts +0 -6
  639. package/dist/src/transfer/test-seraphine.js +0 -105
  640. package/dist/src/transfer/tests/test-store.d.ts +0 -7
  641. package/dist/src/transfer/tests/test-store.js +0 -214
  642. package/dist/src/transfer/types.d.ts +0 -245
  643. package/dist/src/transfer/types.js +0 -6
  644. package/dist/src/types.d.ts +0 -13
  645. package/dist/src/types.js +0 -13
  646. package/dist/src/update/checker.d.ts +0 -34
  647. package/dist/src/update/checker.js +0 -191
  648. package/dist/src/update/executor.d.ts +0 -33
  649. package/dist/src/update/executor.js +0 -217
  650. package/dist/src/update/index.d.ts +0 -33
  651. package/dist/src/update/index.js +0 -64
  652. package/dist/src/update/rate-limiter.d.ts +0 -20
  653. package/dist/src/update/rate-limiter.js +0 -96
  654. package/dist/src/update/validator.d.ts +0 -17
  655. package/dist/src/update/validator.js +0 -123
@@ -1,312 +0,0 @@
1
- /**
2
- * GAIA Adversarial Critic — ADR-135 Track D
3
- *
4
- * After the main agent produces a candidate answer, a Sonnet pass reviews it.
5
- * If the critic verdict is "fail", the orchestrator re-runs the agent with the
6
- * critique injected as additional context.
7
- *
8
- * Motivation (iter 29 finding): tool quality is the bottleneck on L1 (~20.8%).
9
- * The critic catches wrong-because-of-bad-tool-result answers BEFORE submission,
10
- * without requiring better search backends. Expected L1 lift: +3-5pp.
11
- *
12
- * Design:
13
- * - NEW file only; NOT wired into gaia-bench.ts yet to avoid merge conflicts
14
- * with in-flight iter 29/31/34 branches. Wiring is a small follow-up PR.
15
- * - `criticReview()` — single Sonnet call, returns structured verdict.
16
- * - `runGaiaAgentWithCritic()` — orchestration wrapper: runs agent, calls
17
- * critic, retries once on "fail". "uncertain" is treated as "pass" (don't
18
- * burn retries on borderline cases).
19
- * - API errors and malformed JSON from the critic are caught; original
20
- * candidate is returned with an error-flagged verdict.
21
- * - Default opt-in: enableCritic=false. Set true via RunWithCriticOptions.
22
- *
23
- * Cost discipline:
24
- * - Critic uses claude-sonnet-4-6 (separate from the agent's default Haiku).
25
- * - One critic call + one optional retry = max 2 extra Sonnet calls per Q.
26
- * - Approximate extra cost per question: ~$0.003-0.005 (well within budget).
27
- *
28
- * Plugin sync TODO (follow-up PR after gaia-bench wiring):
29
- * - Update plugins/ruflo-workflows/commands/gaia-run.md with --enable-critic flag.
30
- * - Update plugins/ruflo-workflows/skills/gaia-debugging/SKILL.md: add critic
31
- * as a recommended diagnostic step for wrong-answer analysis.
32
- *
33
- * Refs: ADR-135, ADR-133, iter 29 finding, #2156
34
- */
35
- import { execSync } from 'node:child_process';
36
- import { runGaiaAgent, } from './gaia-agent.js';
37
- // ---------------------------------------------------------------------------
38
- // Constants
39
- // ---------------------------------------------------------------------------
40
- const ANTHROPIC_API_URL = 'https://api.anthropic.com/v1/messages';
41
- const ANTHROPIC_API_VERSION = '2023-06-01';
42
- /** Default model for the critic — Sonnet for higher reasoning quality. */
43
- const DEFAULT_CRITIC_MODEL = 'claude-sonnet-4-6';
44
- /** Max tokens for the critic response (verdict JSON is short). */
45
- const CRITIC_MAX_TOKENS = 512;
46
- /** Sonnet pricing (input/output per million tokens, 2026-05-27).
47
- * Used only for cost estimation in CriticVerdict. */
48
- const SONNET_INPUT_COST_PER_M = 3.0;
49
- const SONNET_OUTPUT_COST_PER_M = 15.0;
50
- // ---------------------------------------------------------------------------
51
- // Internal helpers
52
- // ---------------------------------------------------------------------------
53
- /** Resolve Anthropic API key using the same precedence as gaia-agent.ts. */
54
- function resolveApiKey(override) {
55
- if (override)
56
- return override;
57
- const fromEnv = process.env['ANTHROPIC_API_KEY'];
58
- if (fromEnv)
59
- return fromEnv;
60
- try {
61
- return execSync('gcloud secrets versions access latest --secret=ANTHROPIC_API_KEY', { encoding: 'utf8', stdio: ['pipe', 'pipe', 'pipe'] }).trim();
62
- }
63
- catch {
64
- throw new Error('ANTHROPIC_API_KEY not set and gcloud fallback failed. ' +
65
- 'Set ANTHROPIC_API_KEY env var or ensure gcloud access.');
66
- }
67
- }
68
- /**
69
- * Build the critic system prompt.
70
- * Instructs Sonnet to act as an adversarial reviewer and respond in JSON.
71
- */
72
- function buildCriticSystemPrompt() {
73
- return `You are an adversarial reviewer of agent answers on the GAIA benchmark. \
74
- Your job is to find flaws in candidate answers before they are submitted.
75
-
76
- Given a question, a candidate answer, and a summary of the agent's reasoning \
77
- trajectory, you must decide whether the candidate answer is correct.
78
-
79
- Evaluation criteria:
80
- 1. Does the answer directly address what the question asks?
81
- 2. Is there evidence in the trajectory that supports the answer?
82
- 3. Are there obvious flaws (wrong unit, wrong format, missed constraint, \
83
- fabricated source, off-by-one error, truncation)?
84
- 4. For numeric answers: is the magnitude and unit plausible?
85
- 5. For list answers: are all required items present and correctly ordered?
86
-
87
- Respond ONLY with a JSON object — no markdown, no prose outside the JSON:
88
- {"verdict":"pass"|"fail"|"uncertain","reasoning":"<1-3 sentences>","suggestedRevision":"<corrected answer or empty string>"}
89
-
90
- Use "uncertain" only when you genuinely cannot determine correctness from the \
91
- available information. Never use "uncertain" to avoid making a decision when \
92
- evidence is available.`;
93
- }
94
- /**
95
- * Build the critic user message combining the question, candidate answer,
96
- * and a compressed view of the agent trajectory.
97
- */
98
- function buildCriticUserMessage(question, candidateAnswer, trajectory) {
99
- // Summarise trajectory: first 2 + last 2 steps to keep tokens bounded.
100
- const steps = trajectory.steps ?? [];
101
- const summarised = steps.length <= 4
102
- ? steps
103
- : [...steps.slice(0, 2), ...steps.slice(-2)];
104
- const trajectoryText = summarised.length === 0
105
- ? '(no tool calls recorded)'
106
- : summarised.map((s, i) => {
107
- const label = s.tool ? `Step ${i + 1} [${s.tool}]` : `Step ${i + 1}`;
108
- const result = s.result ? ` → ${s.result.slice(0, 200)}` : '';
109
- return `${label}${result}`;
110
- }).join('\n');
111
- return `QUESTION: ${question.question}
112
-
113
- CANDIDATE ANSWER: ${candidateAnswer}
114
-
115
- AGENT TRAJECTORY (${trajectory.turns} turns, summarised):
116
- ${trajectoryText}`;
117
- }
118
- /**
119
- * Attempt to extract a verdict from a potentially malformed JSON string.
120
- * Falls back gracefully to "uncertain" with the raw text preserved.
121
- */
122
- function parseVerdictFallback(raw) {
123
- // Try standard JSON parse first.
124
- try {
125
- const parsed = JSON.parse(raw);
126
- const verdict = parsed['verdict'];
127
- if (verdict === 'pass' || verdict === 'fail' || verdict === 'uncertain') {
128
- return {
129
- verdict,
130
- reasoning: parsed['reasoning'] ?? '',
131
- suggestedRevision: parsed['suggestedRevision'] ?? '',
132
- };
133
- }
134
- }
135
- catch {
136
- // Fall through to regex extraction.
137
- }
138
- // Regex extraction for embedded JSON in prose.
139
- const jsonMatch = raw.match(/\{[\s\S]*?"verdict"\s*:\s*"(pass|fail|uncertain)"[\s\S]*?\}/);
140
- if (jsonMatch) {
141
- try {
142
- const parsed = JSON.parse(jsonMatch[0]);
143
- return {
144
- verdict: parsed['verdict'],
145
- reasoning: parsed['reasoning'] ?? '',
146
- suggestedRevision: parsed['suggestedRevision'] ?? '',
147
- };
148
- }
149
- catch {
150
- // Fall through.
151
- }
152
- }
153
- // Last resort: extract verdict keyword from raw text.
154
- const verdictMatch = raw.match(/\b(pass|fail|uncertain)\b/i);
155
- return {
156
- verdict: verdictMatch
157
- ? verdictMatch[1].toLowerCase()
158
- : 'uncertain',
159
- reasoning: `Critic returned malformed JSON; extracted verdict heuristically.`,
160
- suggestedRevision: '',
161
- rawResponse: raw.slice(0, 500),
162
- };
163
- }
164
- // ---------------------------------------------------------------------------
165
- // Core critic function
166
- // ---------------------------------------------------------------------------
167
- /**
168
- * Run the adversarial critic against a candidate answer.
169
- *
170
- * @param question - The GAIA question being evaluated.
171
- * @param candidateAnswer - The agent's proposed final answer.
172
- * @param trajectory - Lightweight trajectory summary from the agent run.
173
- * @param options - Critic configuration (model, apiKey).
174
- * @returns CriticVerdict with verdict, reasoning, cost.
175
- */
176
- export async function criticReview(question, candidateAnswer, trajectory, options) {
177
- const model = options?.model ?? DEFAULT_CRITIC_MODEL;
178
- const apiKey = resolveApiKey(options?.apiKey);
179
- const t0 = Date.now();
180
- const requestBody = {
181
- model,
182
- max_tokens: CRITIC_MAX_TOKENS,
183
- system: buildCriticSystemPrompt(),
184
- messages: [
185
- {
186
- role: 'user',
187
- content: buildCriticUserMessage(question, candidateAnswer, trajectory),
188
- },
189
- ],
190
- };
191
- let rawResponseText = '';
192
- try {
193
- const response = await fetch(ANTHROPIC_API_URL, {
194
- method: 'POST',
195
- headers: {
196
- 'Content-Type': 'application/json',
197
- 'x-api-key': apiKey,
198
- 'anthropic-version': ANTHROPIC_API_VERSION,
199
- },
200
- body: JSON.stringify(requestBody),
201
- });
202
- if (!response.ok) {
203
- const errText = await response.text().catch(() => '');
204
- throw new Error(`Anthropic API error ${response.status}: ${errText.slice(0, 200)}`);
205
- }
206
- const data = await response.json();
207
- const inputTokens = data.usage?.input_tokens ?? 0;
208
- const outputTokens = data.usage?.output_tokens ?? 0;
209
- const costUsd = (inputTokens / 1_000_000) * SONNET_INPUT_COST_PER_M +
210
- (outputTokens / 1_000_000) * SONNET_OUTPUT_COST_PER_M;
211
- rawResponseText = data.content
212
- .filter(b => b.type === 'text')
213
- .map(b => b.text ?? '')
214
- .join('');
215
- const parsed = parseVerdictFallback(rawResponseText);
216
- return {
217
- verdict: parsed.verdict ?? 'uncertain',
218
- reasoning: parsed.reasoning ?? '',
219
- suggestedRevision: parsed.suggestedRevision ?? '',
220
- costUsd,
221
- ...(parsed.rawResponse ? { rawResponse: parsed.rawResponse } : {}),
222
- };
223
- }
224
- catch (err) {
225
- const wallMs = Date.now() - t0;
226
- const errMsg = err instanceof Error ? err.message : String(err);
227
- // Graceful fallback: treat critic error as "uncertain" so agent result
228
- // is still returned rather than throwing.
229
- return {
230
- verdict: 'uncertain',
231
- reasoning: `Critic call failed after ${wallMs}ms: ${errMsg.slice(0, 200)}`,
232
- suggestedRevision: '',
233
- costUsd: 0,
234
- error: true,
235
- rawResponse: rawResponseText.slice(0, 200) || undefined,
236
- };
237
- }
238
- }
239
- // ---------------------------------------------------------------------------
240
- // Orchestration wrapper
241
- // ---------------------------------------------------------------------------
242
- /**
243
- * Run the GAIA agent with an optional adversarial critic pass.
244
- *
245
- * When enableCritic=false (default), this is a thin pass-through to
246
- * runGaiaAgent with an empty criticVerdicts array.
247
- *
248
- * When enableCritic=true:
249
- * 1. Run runGaiaAgent normally.
250
- * 2. If the agent produced a finalAnswer, call criticReview.
251
- * 3. If verdict is "fail" and retriesAttempted < maxRetries:
252
- * a. Re-run the agent with the critique injected as additional context.
253
- * b. Call criticReview again on the new answer.
254
- * 4. Return the final result with all critic verdicts attached.
255
- *
256
- * Note on "uncertain": treated as "pass" (no retry triggered).
257
- */
258
- export async function runGaiaAgentWithCritic(question, options = {}) {
259
- const { enableCritic = false, criticOptions, ...agentOptions } = options;
260
- // Fast path: critic disabled.
261
- if (!enableCritic) {
262
- const result = await runGaiaAgent(question, agentOptions);
263
- return { ...result, criticVerdicts: [], retriesAttempted: 0 };
264
- }
265
- const maxRetries = criticOptions?.maxRetries ?? 1;
266
- const criticVerdicts = [];
267
- let retriesAttempted = 0;
268
- // First agent run.
269
- let agentResult = await runGaiaAgent(question, agentOptions);
270
- // If agent timed out or errored with no answer, skip critic.
271
- if (agentResult.finalAnswer == null) {
272
- return { ...agentResult, criticVerdicts, retriesAttempted };
273
- }
274
- // Build a lightweight trajectory summary from the agent result.
275
- // gaia-agent.ts doesn't expose step-level detail in GaiaAgentResult, so we
276
- // synthesise from the available fields (tool call counts, turn count).
277
- const makeTrajectory = (result) => ({
278
- turns: result.turns,
279
- steps: Object.entries(result.toolCallsByName).map(([tool, count]) => ({
280
- tool,
281
- result: `called ${count} time(s)`,
282
- })),
283
- });
284
- // First critic pass.
285
- let verdict = await criticReview(question, agentResult.finalAnswer, makeTrajectory(agentResult), criticOptions);
286
- criticVerdicts.push(verdict);
287
- // Retry loop on "fail".
288
- while (verdict.verdict === 'fail' && retriesAttempted < maxRetries) {
289
- retriesAttempted += 1;
290
- // Inject critique as additional context via the system prompt append.
291
- // We pass it through GaiaAgentOptions.criticFeedback — gaia-agent.ts does
292
- // NOT yet read this field (wiring is follow-up PR), but storing it here
293
- // makes the contract explicit and harmless (unknown options are ignored).
294
- const retryOptions = {
295
- ...agentOptions,
296
- criticFeedback: `Previous answer "${agentResult.finalAnswer}" was flagged by adversarial review: ` +
297
- `${verdict.reasoning}` +
298
- (verdict.suggestedRevision
299
- ? ` Suggested revision: ${verdict.suggestedRevision}`
300
- : ''),
301
- };
302
- agentResult = await runGaiaAgent(question, retryOptions);
303
- if (agentResult.finalAnswer == null) {
304
- // Retry yielded no answer; stop retrying.
305
- break;
306
- }
307
- verdict = await criticReview(question, agentResult.finalAnswer, makeTrajectory(agentResult), criticOptions);
308
- criticVerdicts.push(verdict);
309
- }
310
- return { ...agentResult, criticVerdicts, retriesAttempted };
311
- }
312
- //# sourceMappingURL=gaia-critic.js.map
@@ -1,21 +0,0 @@
1
- /**
2
- * GAIA Critic Smoke Tests — ADR-135 Track D
3
- *
4
- * Tests for the adversarial critic agent in gaia-critic.ts.
5
- * All tests use mocked responses — NO live API calls.
6
- *
7
- * Test coverage:
8
- * 1. Critic returns "pass" → no retry, immediately returns candidate
9
- * 2. Critic returns "fail" with suggestedRevision → triggers one retry
10
- * 3. Critic returns "fail" twice → retries exhausted, returns last candidate
11
- * 4. Critic returns "uncertain" → treated as "pass", no retry
12
- * 5. API error in critic → graceful fallback, returns candidate as-is
13
- * 6. Malformed JSON from critic → fallback parser extracts verdict
14
- *
15
- * Usage (no API key required):
16
- * npx tsx src/benchmarks/gaia-critic.smoke.ts
17
- *
18
- * Refs: ADR-135, #2156
19
- */
20
- export {};
21
- //# sourceMappingURL=gaia-critic.smoke.d.ts.map
@@ -1,327 +0,0 @@
1
- /**
2
- * GAIA Critic Smoke Tests — ADR-135 Track D
3
- *
4
- * Tests for the adversarial critic agent in gaia-critic.ts.
5
- * All tests use mocked responses — NO live API calls.
6
- *
7
- * Test coverage:
8
- * 1. Critic returns "pass" → no retry, immediately returns candidate
9
- * 2. Critic returns "fail" with suggestedRevision → triggers one retry
10
- * 3. Critic returns "fail" twice → retries exhausted, returns last candidate
11
- * 4. Critic returns "uncertain" → treated as "pass", no retry
12
- * 5. API error in critic → graceful fallback, returns candidate as-is
13
- * 6. Malformed JSON from critic → fallback parser extracts verdict
14
- *
15
- * Usage (no API key required):
16
- * npx tsx src/benchmarks/gaia-critic.smoke.ts
17
- *
18
- * Refs: ADR-135, #2156
19
- */
20
- import { criticReview, runGaiaAgentWithCritic, } from './gaia-critic.js';
21
- // ---------------------------------------------------------------------------
22
- // Test helpers
23
- // ---------------------------------------------------------------------------
24
- let passed = 0;
25
- let failed = 0;
26
- function assert(condition, message) {
27
- if (condition) {
28
- console.log(` PASS ${message}`);
29
- passed++;
30
- }
31
- else {
32
- console.error(` FAIL ${message}`);
33
- failed++;
34
- }
35
- }
36
- function assertEqual(actual, expected, message) {
37
- assert(actual === expected, `${message} (expected ${String(expected)}, got ${String(actual)})`);
38
- }
39
- /** A minimal GaiaQuestion fixture. */
40
- const FIXTURE_QUESTION = {
41
- task_id: 'smoke-001',
42
- level: 1,
43
- question: 'What is the capital of France?',
44
- final_answer: 'Paris',
45
- file_name: null,
46
- file_path: null,
47
- };
48
- /** Minimal GaiaAgentResult with a given finalAnswer. */
49
- function makeAgentResult(finalAnswer, turns = 2) {
50
- return {
51
- questionId: FIXTURE_QUESTION.task_id,
52
- finalAnswer,
53
- turns,
54
- toolCallsByName: { web_search: 1 },
55
- totalInputTokens: 100,
56
- totalOutputTokens: 50,
57
- wallMs: 1200,
58
- };
59
- }
60
- function installFetchMock(responses) {
61
- let callIndex = 0;
62
- const original = globalThis.fetch;
63
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
64
- globalThis.fetch = async (_url, _init) => {
65
- const resp = responses[Math.min(callIndex, responses.length - 1)];
66
- callIndex++;
67
- if (!resp.ok) {
68
- return {
69
- ok: false,
70
- status: resp.status ?? 500,
71
- text: async () => resp.text ?? 'Internal error',
72
- json: async () => { throw new Error('not ok'); },
73
- };
74
- }
75
- return {
76
- ok: true,
77
- status: 200,
78
- json: async () => resp.body ?? {},
79
- text: async () => JSON.stringify(resp.body ?? {}),
80
- };
81
- };
82
- return () => { globalThis.fetch = original; };
83
- }
84
- /**
85
- * Build a mock Anthropic response body containing a critic verdict JSON.
86
- */
87
- function mockCriticResponse(verdict, reasoning, suggestedRevision = '') {
88
- const content = JSON.stringify({ verdict, reasoning, suggestedRevision });
89
- return {
90
- content: [{ type: 'text', text: content }],
91
- usage: { input_tokens: 200, output_tokens: 40 },
92
- };
93
- }
94
- /**
95
- * Mock for runGaiaAgent that immediately returns a fixed result.
96
- * We monkey-patch the module-level import by wrapping runGaiaAgentWithCritic
97
- * via its options.catalogue approach — but since gaia-agent's runGaiaAgent is
98
- * imported directly, we mock at the fetch level instead (the agent also calls
99
- * the Anthropic API, so we intercept there).
100
- *
101
- * For the smoke tests we only exercise criticReview directly in Tests 1-6, and
102
- * use a simplified version of runGaiaAgentWithCritic that pre-supplies a
103
- * mocked agent result rather than actually calling the API for the agent run.
104
- *
105
- * This avoids needing to rewrite the import mechanism just for smoke tests.
106
- */
107
- // ---------------------------------------------------------------------------
108
- // Test 1: critic returns "pass" → no retry
109
- // ---------------------------------------------------------------------------
110
- async function test1_criticPass() {
111
- console.log('\nTest 1: critic returns "pass" → no retry');
112
- const restore = installFetchMock([
113
- { ok: true, body: mockCriticResponse('pass', 'Answer is correct.') },
114
- ]);
115
- try {
116
- const verdict = await criticReview(FIXTURE_QUESTION, 'Paris', { steps: [{ tool: 'web_search', result: 'Paris is the capital' }], turns: 2 }, { model: 'claude-sonnet-4-6', apiKey: 'test-key' });
117
- assertEqual(verdict.verdict, 'pass', 'verdict is "pass"');
118
- assert(verdict.reasoning.length > 0, 'reasoning is non-empty');
119
- assert(verdict.costUsd >= 0, 'costUsd is non-negative');
120
- assert(!verdict.error, 'no error flag');
121
- }
122
- finally {
123
- restore();
124
- }
125
- }
126
- // ---------------------------------------------------------------------------
127
- // Test 2: critic returns "fail" with suggestedRevision
128
- // ---------------------------------------------------------------------------
129
- async function test2_criticFail() {
130
- console.log('\nTest 2: critic returns "fail" with suggestedRevision');
131
- const restore = installFetchMock([
132
- {
133
- ok: true,
134
- body: mockCriticResponse('fail', 'The answer is the wrong city.', 'Paris'),
135
- },
136
- ]);
137
- try {
138
- const verdict = await criticReview(FIXTURE_QUESTION, 'Lyon', { steps: [{ tool: 'web_search', result: 'Lyon is in France' }], turns: 1 }, { model: 'claude-sonnet-4-6', apiKey: 'test-key' });
139
- assertEqual(verdict.verdict, 'fail', 'verdict is "fail"');
140
- assert((verdict.suggestedRevision ?? '').length > 0, 'suggestedRevision is non-empty');
141
- assertEqual(verdict.suggestedRevision, 'Paris', 'suggestedRevision is "Paris"');
142
- }
143
- finally {
144
- restore();
145
- }
146
- }
147
- // ---------------------------------------------------------------------------
148
- // Test 3: critic returns "fail" twice → retries exhausted
149
- // ---------------------------------------------------------------------------
150
- async function test3_retriesExhausted() {
151
- console.log('\nTest 3: critic fails twice → retries exhausted, returns last candidate');
152
- // We test this at the runGaiaAgentWithCritic level.
153
- // We need to mock BOTH the critic fetch calls AND the agent API calls.
154
- // Strategy: sequence the mock responses in the order they will be called.
155
- //
156
- // Call sequence (with enableCritic=true, maxRetries=1):
157
- // 1. runGaiaAgent attempt 1 → agent Anthropic API call (returns answer "Lyon")
158
- // 2. criticReview attempt 1 → critic API call (returns "fail")
159
- // 3. runGaiaAgent attempt 2 → agent Anthropic API call (returns answer "Marseille")
160
- // 4. criticReview attempt 2 → critic API call (returns "fail")
161
- //
162
- // For simplicity we make the agent calls also return valid Anthropic responses
163
- // that produce a FINAL_ANSWER (the agent code parses stop_reason=end_turn).
164
- const agentResponseLyon = {
165
- id: 'msg_01',
166
- type: 'message',
167
- role: 'assistant',
168
- stop_reason: 'end_turn',
169
- content: [{ type: 'text', text: 'FINAL_ANSWER: Lyon' }],
170
- usage: { input_tokens: 150, output_tokens: 20 },
171
- model: 'claude-haiku-4-5',
172
- };
173
- const agentResponseMarseille = {
174
- id: 'msg_02',
175
- type: 'message',
176
- role: 'assistant',
177
- stop_reason: 'end_turn',
178
- content: [{ type: 'text', text: 'FINAL_ANSWER: Marseille' }],
179
- usage: { input_tokens: 150, output_tokens: 20 },
180
- model: 'claude-haiku-4-5',
181
- };
182
- const restore = installFetchMock([
183
- { ok: true, body: agentResponseLyon }, // agent call 1
184
- { ok: true, body: mockCriticResponse('fail', 'Wrong city.', 'Paris') }, // critic 1
185
- { ok: true, body: agentResponseMarseille }, // agent call 2 (retry)
186
- { ok: true, body: mockCriticResponse('fail', 'Still wrong.', 'Paris') }, // critic 2
187
- ]);
188
- try {
189
- const result = await runGaiaAgentWithCritic(FIXTURE_QUESTION, {
190
- enableCritic: true,
191
- apiKey: 'test-key',
192
- criticOptions: { apiKey: 'test-key', maxRetries: 1 },
193
- });
194
- assertEqual(result.retriesAttempted, 1, 'retriesAttempted is 1');
195
- assertEqual(result.criticVerdicts.length, 2, 'two critic verdicts collected');
196
- assertEqual(result.criticVerdicts[0].verdict, 'fail', 'first verdict is fail');
197
- assertEqual(result.criticVerdicts[1].verdict, 'fail', 'second verdict is fail');
198
- // Last candidate is returned regardless.
199
- assert(result.finalAnswer !== null, 'finalAnswer is non-null (last candidate returned)');
200
- }
201
- finally {
202
- restore();
203
- }
204
- }
205
- // ---------------------------------------------------------------------------
206
- // Test 4: critic returns "uncertain" → treated as "pass", no retry
207
- // ---------------------------------------------------------------------------
208
- async function test4_uncertainAsPass() {
209
- console.log('\nTest 4: critic returns "uncertain" → treated as pass, no retry');
210
- const restore = installFetchMock([
211
- {
212
- ok: true,
213
- body: mockCriticResponse('uncertain', 'Cannot verify without more context.'),
214
- },
215
- ]);
216
- try {
217
- const verdict = await criticReview(FIXTURE_QUESTION, 'Paris', { steps: [], turns: 1 }, { model: 'claude-sonnet-4-6', apiKey: 'test-key' });
218
- assertEqual(verdict.verdict, 'uncertain', 'verdict is "uncertain"');
219
- // Verify orchestrator treats uncertain as pass (no retry fired).
220
- // We simulate by checking the logic directly: uncertain !== 'fail', so loop
221
- // body is skipped. We test it via runGaiaAgentWithCritic with a minimal
222
- // agent mock that returns a final answer.
223
- const agentResponseParis = {
224
- id: 'msg_03',
225
- type: 'message',
226
- role: 'assistant',
227
- stop_reason: 'end_turn',
228
- content: [{ type: 'text', text: 'FINAL_ANSWER: Paris' }],
229
- usage: { input_tokens: 100, output_tokens: 15 },
230
- model: 'claude-haiku-4-5',
231
- };
232
- const restore2 = installFetchMock([
233
- { ok: true, body: agentResponseParis }, // agent call
234
- { ok: true, body: mockCriticResponse('uncertain', 'Cannot verify.') }, // critic
235
- ]);
236
- try {
237
- const result = await runGaiaAgentWithCritic(FIXTURE_QUESTION, {
238
- enableCritic: true,
239
- apiKey: 'test-key',
240
- criticOptions: { apiKey: 'test-key', maxRetries: 1 },
241
- });
242
- assertEqual(result.retriesAttempted, 0, 'no retries for uncertain verdict');
243
- assertEqual(result.criticVerdicts.length, 1, 'one critic verdict collected');
244
- assertEqual(result.criticVerdicts[0].verdict, 'uncertain', 'verdict is uncertain');
245
- }
246
- finally {
247
- restore2();
248
- }
249
- }
250
- finally {
251
- restore();
252
- }
253
- }
254
- // ---------------------------------------------------------------------------
255
- // Test 5: API error in critic → graceful fallback
256
- // ---------------------------------------------------------------------------
257
- async function test5_apiErrorFallback() {
258
- console.log('\nTest 5: API error in critic → graceful fallback, original candidate returned');
259
- const restore = installFetchMock([
260
- { ok: false, status: 529, text: 'Overloaded' },
261
- ]);
262
- try {
263
- const verdict = await criticReview(FIXTURE_QUESTION, 'Paris', { steps: [], turns: 1 }, { model: 'claude-sonnet-4-6', apiKey: 'test-key' });
264
- // On API error, critic returns uncertain with error flag — does not throw.
265
- assertEqual(verdict.verdict, 'uncertain', 'verdict is "uncertain" on API error');
266
- assert(verdict.error === true, 'error flag is set');
267
- assert(verdict.costUsd === 0, 'costUsd is 0 on error');
268
- assert(verdict.reasoning.includes('failed'), 'reasoning mentions failure');
269
- }
270
- finally {
271
- restore();
272
- }
273
- }
274
- // ---------------------------------------------------------------------------
275
- // Test 6: malformed JSON from critic → fallback parser
276
- // ---------------------------------------------------------------------------
277
- async function test6_malformedJson() {
278
- console.log('\nTest 6: malformed JSON from critic → fallback parser extracts verdict');
279
- // Simulate Sonnet returning prose with embedded JSON fragment.
280
- const malformedBody = {
281
- content: [{
282
- type: 'text',
283
- text: 'After careful review, I believe the answer is wrong. {"verdict":"fail","reasoning":"Incorrect city","suggestedRevision":"Paris"} The agent should try again.',
284
- }],
285
- usage: { input_tokens: 180, output_tokens: 60 },
286
- };
287
- const restore = installFetchMock([
288
- { ok: true, body: malformedBody },
289
- ]);
290
- try {
291
- const verdict = await criticReview(FIXTURE_QUESTION, 'Lyon', { steps: [], turns: 1 }, { model: 'claude-sonnet-4-6', apiKey: 'test-key' });
292
- // Fallback parser should extract "fail" from the embedded JSON.
293
- assertEqual(verdict.verdict, 'fail', 'fallback parser extracts "fail" verdict');
294
- assert(!verdict.error, 'no error flag for recoverable parse');
295
- }
296
- finally {
297
- restore();
298
- }
299
- }
300
- // ---------------------------------------------------------------------------
301
- // Run all tests
302
- // ---------------------------------------------------------------------------
303
- async function main() {
304
- console.log('=== GAIA Critic Smoke Tests (ADR-135 Track D) ===');
305
- console.log('All tests use mocked responses — no live API calls.\n');
306
- try {
307
- await test1_criticPass();
308
- await test2_criticFail();
309
- await test3_retriesExhausted();
310
- await test4_uncertainAsPass();
311
- await test5_apiErrorFallback();
312
- await test6_malformedJson();
313
- }
314
- catch (err) {
315
- console.error('\nUnexpected test runner error:', err);
316
- process.exit(1);
317
- }
318
- console.log(`\n=== Results: ${passed} passed, ${failed} failed ===`);
319
- if (failed > 0) {
320
- process.exit(1);
321
- }
322
- }
323
- main().catch(err => {
324
- console.error(err);
325
- process.exit(1);
326
- });
327
- //# sourceMappingURL=gaia-critic.smoke.js.map