agent-nuvira 3.3.3 → 3.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (366) hide show
  1. package/README.md +13 -5
  2. package/dist/agent-sdk/src/agent.d.ts +2 -0
  3. package/dist/agent-sdk/src/agent.d.ts.map +1 -1
  4. package/dist/agent-sdk/src/define.d.ts +64 -0
  5. package/dist/agent-sdk/src/define.d.ts.map +1 -0
  6. package/dist/agent-sdk/src/define.js +76 -0
  7. package/dist/agent-sdk/src/define.js.map +1 -0
  8. package/dist/agent-sdk/src/index.d.ts +9 -0
  9. package/dist/agent-sdk/src/index.d.ts.map +1 -1
  10. package/dist/agent-sdk/src/index.js +9 -0
  11. package/dist/agent-sdk/src/index.js.map +1 -1
  12. package/dist/agent-sdk/src/scaffold.d.ts +11 -0
  13. package/dist/agent-sdk/src/scaffold.d.ts.map +1 -1
  14. package/dist/agent-sdk/src/scaffold.js +16 -6
  15. package/dist/agent-sdk/src/scaffold.js.map +1 -1
  16. package/dist/agents/long-form-plan.d.ts.map +1 -1
  17. package/dist/agents/long-form-plan.js +2 -1
  18. package/dist/agents/long-form-plan.js.map +1 -1
  19. package/dist/agents/orchestrator.d.ts.map +1 -1
  20. package/dist/agents/orchestrator.js +5 -4
  21. package/dist/agents/orchestrator.js.map +1 -1
  22. package/dist/cli/agent.d.ts +2 -2
  23. package/dist/cli/agent.js +10 -10
  24. package/dist/cli/chat.d.ts +94 -0
  25. package/dist/cli/chat.d.ts.map +1 -1
  26. package/dist/cli/chat.js +354 -20
  27. package/dist/cli/chat.js.map +1 -1
  28. package/dist/cli/cli-program.d.ts.map +1 -1
  29. package/dist/cli/cli-program.js +5 -0
  30. package/dist/cli/cli-program.js.map +1 -1
  31. package/dist/cli/config.d.ts.map +1 -1
  32. package/dist/cli/config.js +9 -1
  33. package/dist/cli/config.js.map +1 -1
  34. package/dist/cli/doctor.d.ts.map +1 -1
  35. package/dist/cli/doctor.js +3 -2
  36. package/dist/cli/doctor.js.map +1 -1
  37. package/dist/cli/edit.js +2 -2
  38. package/dist/cli/eval.d.ts +17 -0
  39. package/dist/cli/eval.d.ts.map +1 -1
  40. package/dist/cli/eval.js +105 -2
  41. package/dist/cli/eval.js.map +1 -1
  42. package/dist/cli/execute.d.ts +16 -1
  43. package/dist/cli/execute.d.ts.map +1 -1
  44. package/dist/cli/execute.js +192 -22
  45. package/dist/cli/execute.js.map +1 -1
  46. package/dist/cli/loop-executor.d.ts +55 -0
  47. package/dist/cli/loop-executor.d.ts.map +1 -1
  48. package/dist/cli/loop-executor.js +212 -13
  49. package/dist/cli/loop-executor.js.map +1 -1
  50. package/dist/cli/model.d.ts.map +1 -1
  51. package/dist/cli/model.js +9 -8
  52. package/dist/cli/model.js.map +1 -1
  53. package/dist/cli/models.d.ts +1 -1
  54. package/dist/cli/models.js +4 -4
  55. package/dist/cli/parity.d.ts +85 -0
  56. package/dist/cli/parity.d.ts.map +1 -0
  57. package/dist/cli/parity.js +506 -0
  58. package/dist/cli/parity.js.map +1 -0
  59. package/dist/cli/plan.d.ts.map +1 -1
  60. package/dist/cli/plan.js +2 -1
  61. package/dist/cli/plan.js.map +1 -1
  62. package/dist/cli/retrieval.d.ts.map +1 -1
  63. package/dist/cli/retrieval.js +5 -4
  64. package/dist/cli/retrieval.js.map +1 -1
  65. package/dist/cli/sdk.js +4 -4
  66. package/dist/cli/sdk.js.map +1 -1
  67. package/dist/cli/trace.d.ts.map +1 -1
  68. package/dist/cli/trace.js +2 -1
  69. package/dist/cli/trace.js.map +1 -1
  70. package/dist/cli/workflow.js +2 -2
  71. package/dist/cli/workflow.js.map +1 -1
  72. package/dist/config/process-env.d.ts +136 -0
  73. package/dist/config/process-env.d.ts.map +1 -0
  74. package/dist/config/process-env.js +217 -0
  75. package/dist/config/process-env.js.map +1 -0
  76. package/dist/config/types.d.ts +44 -0
  77. package/dist/config/types.d.ts.map +1 -1
  78. package/dist/findings/verdicts.d.ts +241 -0
  79. package/dist/findings/verdicts.d.ts.map +1 -0
  80. package/dist/findings/verdicts.js +284 -0
  81. package/dist/findings/verdicts.js.map +1 -0
  82. package/dist/gateway/adapters.d.ts +65 -0
  83. package/dist/gateway/adapters.d.ts.map +1 -1
  84. package/dist/gateway/adapters.js +216 -10
  85. package/dist/gateway/adapters.js.map +1 -1
  86. package/dist/gateway/channel-directory.d.ts +31 -0
  87. package/dist/gateway/channel-directory.d.ts.map +1 -1
  88. package/dist/gateway/channel-directory.js +40 -0
  89. package/dist/gateway/channel-directory.js.map +1 -1
  90. package/dist/gateway/gateway-log.d.ts +1 -1
  91. package/dist/gateway/gateway-log.d.ts.map +1 -1
  92. package/dist/gateway/gateway-log.js.map +1 -1
  93. package/dist/gateway/hooks.d.ts +87 -19
  94. package/dist/gateway/hooks.d.ts.map +1 -1
  95. package/dist/gateway/hooks.js +62 -23
  96. package/dist/gateway/hooks.js.map +1 -1
  97. package/dist/gateway/inbound-media.d.ts +147 -0
  98. package/dist/gateway/inbound-media.d.ts.map +1 -0
  99. package/dist/gateway/inbound-media.js +317 -0
  100. package/dist/gateway/inbound-media.js.map +1 -0
  101. package/dist/gateway/inbox.d.ts +8 -1
  102. package/dist/gateway/inbox.d.ts.map +1 -1
  103. package/dist/gateway/inbox.js.map +1 -1
  104. package/dist/gateway/platform-config.d.ts +14 -0
  105. package/dist/gateway/platform-config.d.ts.map +1 -1
  106. package/dist/gateway/platform-config.js +26 -8
  107. package/dist/gateway/platform-config.js.map +1 -1
  108. package/dist/gateway/realtime.d.ts +114 -0
  109. package/dist/gateway/realtime.d.ts.map +1 -0
  110. package/dist/gateway/realtime.js +402 -0
  111. package/dist/gateway/realtime.js.map +1 -0
  112. package/dist/gateway/registry.d.ts +31 -0
  113. package/dist/gateway/registry.d.ts.map +1 -1
  114. package/dist/gateway/registry.js +224 -4
  115. package/dist/gateway/registry.js.map +1 -1
  116. package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
  117. package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
  118. package/dist/gateway/whatsapp/baileys-bridge.js +123 -4
  119. package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
  120. package/dist/gateway/whatsapp/bridge.d.ts +6 -2
  121. package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
  122. package/dist/gateway/whatsapp/bridge.js.map +1 -1
  123. package/dist/index.js +8 -0
  124. package/dist/index.js.map +1 -1
  125. package/dist/inference/factory.d.ts +14 -0
  126. package/dist/inference/factory.d.ts.map +1 -1
  127. package/dist/inference/factory.js +17 -0
  128. package/dist/inference/factory.js.map +1 -1
  129. package/dist/inference/groq-adapter.d.ts +2 -0
  130. package/dist/inference/groq-adapter.d.ts.map +1 -1
  131. package/dist/inference/groq-adapter.js +16 -6
  132. package/dist/inference/groq-adapter.js.map +1 -1
  133. package/dist/inference/tools.d.ts.map +1 -1
  134. package/dist/inference/tools.js +29 -0
  135. package/dist/inference/tools.js.map +1 -1
  136. package/dist/learning/benchmark.d.ts.map +1 -1
  137. package/dist/learning/benchmark.js +3 -2
  138. package/dist/learning/benchmark.js.map +1 -1
  139. package/dist/learning/continuation.d.ts.map +1 -1
  140. package/dist/learning/continuation.js +2 -1
  141. package/dist/learning/continuation.js.map +1 -1
  142. package/dist/learning/cost-tracker.d.ts.map +1 -1
  143. package/dist/learning/cost-tracker.js +2 -1
  144. package/dist/learning/cost-tracker.js.map +1 -1
  145. package/dist/learning/deferred-task.d.ts.map +1 -1
  146. package/dist/learning/deferred-task.js +13 -4
  147. package/dist/learning/deferred-task.js.map +1 -1
  148. package/dist/learning/eval-framework.d.ts.map +1 -1
  149. package/dist/learning/eval-framework.js +2 -1
  150. package/dist/learning/eval-framework.js.map +1 -1
  151. package/dist/learning/long-form.d.ts.map +1 -1
  152. package/dist/learning/long-form.js +2 -1
  153. package/dist/learning/long-form.js.map +1 -1
  154. package/dist/learning/model-reachability.d.ts +95 -0
  155. package/dist/learning/model-reachability.d.ts.map +1 -0
  156. package/dist/learning/model-reachability.js +111 -0
  157. package/dist/learning/model-reachability.js.map +1 -0
  158. package/dist/learning/model-registry.d.ts +22 -0
  159. package/dist/learning/model-registry.d.ts.map +1 -1
  160. package/dist/learning/model-registry.js +26 -1
  161. package/dist/learning/model-registry.js.map +1 -1
  162. package/dist/learning/model-verify-job.d.ts +131 -0
  163. package/dist/learning/model-verify-job.d.ts.map +1 -0
  164. package/dist/learning/model-verify-job.js +221 -0
  165. package/dist/learning/model-verify-job.js.map +1 -0
  166. package/dist/learning/reasoning-cache.d.ts.map +1 -1
  167. package/dist/learning/reasoning-cache.js +2 -1
  168. package/dist/learning/reasoning-cache.js.map +1 -1
  169. package/dist/learning/reasoning-trace.d.ts +37 -1
  170. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  171. package/dist/learning/reasoning-trace.js +66 -0
  172. package/dist/learning/reasoning-trace.js.map +1 -1
  173. package/dist/learning/resilient-call.d.ts.map +1 -1
  174. package/dist/learning/resilient-call.js +2 -1
  175. package/dist/learning/resilient-call.js.map +1 -1
  176. package/dist/learning/retrieval.d.ts.map +1 -1
  177. package/dist/learning/retrieval.js +2 -1
  178. package/dist/learning/retrieval.js.map +1 -1
  179. package/dist/learning/seeded-benchmark.d.ts +160 -0
  180. package/dist/learning/seeded-benchmark.d.ts.map +1 -0
  181. package/dist/learning/seeded-benchmark.js +321 -0
  182. package/dist/learning/seeded-benchmark.js.map +1 -0
  183. package/dist/learning/seeded-bugs.d.ts +142 -0
  184. package/dist/learning/seeded-bugs.d.ts.map +1 -0
  185. package/dist/learning/seeded-bugs.js +535 -0
  186. package/dist/learning/seeded-bugs.js.map +1 -0
  187. package/dist/learning/step-checkpoint.d.ts +127 -0
  188. package/dist/learning/step-checkpoint.d.ts.map +1 -0
  189. package/dist/learning/step-checkpoint.js +244 -0
  190. package/dist/learning/step-checkpoint.js.map +1 -0
  191. package/dist/nlu/intent-confirm.d.ts +23 -1
  192. package/dist/nlu/intent-confirm.d.ts.map +1 -1
  193. package/dist/nlu/intent-confirm.js +85 -2
  194. package/dist/nlu/intent-confirm.js.map +1 -1
  195. package/dist/nlu/learnings.d.ts +7 -0
  196. package/dist/nlu/learnings.d.ts.map +1 -1
  197. package/dist/nlu/learnings.js +7 -0
  198. package/dist/nlu/learnings.js.map +1 -1
  199. package/dist/observability/debug-log.d.ts +250 -0
  200. package/dist/observability/debug-log.d.ts.map +1 -0
  201. package/dist/observability/debug-log.js +500 -0
  202. package/dist/observability/debug-log.js.map +1 -0
  203. package/dist/observability/event-bus.d.ts.map +1 -1
  204. package/dist/observability/event-bus.js +4 -1
  205. package/dist/observability/event-bus.js.map +1 -1
  206. package/dist/observability/otel.d.ts +278 -0
  207. package/dist/observability/otel.d.ts.map +1 -0
  208. package/dist/observability/otel.js +590 -0
  209. package/dist/observability/otel.js.map +1 -0
  210. package/dist/parity/drivers.d.ts +99 -0
  211. package/dist/parity/drivers.d.ts.map +1 -0
  212. package/dist/parity/drivers.js +1362 -0
  213. package/dist/parity/drivers.js.map +1 -0
  214. package/dist/parity/graph.d.ts +73 -0
  215. package/dist/parity/graph.d.ts.map +1 -0
  216. package/dist/parity/graph.js +162 -0
  217. package/dist/parity/graph.js.map +1 -0
  218. package/dist/parity/matrix.d.ts +105 -0
  219. package/dist/parity/matrix.d.ts.map +1 -0
  220. package/dist/parity/matrix.js +352 -0
  221. package/dist/parity/matrix.js.map +1 -0
  222. package/dist/parity/observation.d.ts +444 -0
  223. package/dist/parity/observation.d.ts.map +1 -0
  224. package/dist/parity/observation.js +333 -0
  225. package/dist/parity/observation.js.map +1 -0
  226. package/dist/parity/scenarios.d.ts +229 -0
  227. package/dist/parity/scenarios.d.ts.map +1 -0
  228. package/dist/parity/scenarios.js +175 -0
  229. package/dist/parity/scenarios.js.map +1 -0
  230. package/dist/parity/surfaces.d.ts +122 -0
  231. package/dist/parity/surfaces.d.ts.map +1 -0
  232. package/dist/parity/surfaces.js +190 -0
  233. package/dist/parity/surfaces.js.map +1 -0
  234. package/dist/runtime/fault-injection.d.ts +173 -0
  235. package/dist/runtime/fault-injection.d.ts.map +1 -0
  236. package/dist/runtime/fault-injection.js +281 -0
  237. package/dist/runtime/fault-injection.js.map +1 -0
  238. package/dist/tools/child-agent-entry.d.ts +23 -0
  239. package/dist/tools/child-agent-entry.d.ts.map +1 -0
  240. package/dist/tools/child-agent-entry.js +129 -0
  241. package/dist/tools/child-agent-entry.js.map +1 -0
  242. package/dist/tools/child-agent-runtime.d.ts +124 -0
  243. package/dist/tools/child-agent-runtime.d.ts.map +1 -0
  244. package/dist/tools/child-agent-runtime.js +704 -0
  245. package/dist/tools/child-agent-runtime.js.map +1 -0
  246. package/dist/tools/coding-tools.d.ts.map +1 -1
  247. package/dist/tools/coding-tools.js +82 -10
  248. package/dist/tools/coding-tools.js.map +1 -1
  249. package/dist/tools/delegation-system.d.ts +31 -0
  250. package/dist/tools/delegation-system.d.ts.map +1 -1
  251. package/dist/tools/delegation-system.js +70 -9
  252. package/dist/tools/delegation-system.js.map +1 -1
  253. package/dist/tools/extract/docx.d.ts +27 -0
  254. package/dist/tools/extract/docx.d.ts.map +1 -0
  255. package/dist/tools/extract/docx.js +48 -0
  256. package/dist/tools/extract/docx.js.map +1 -0
  257. package/dist/tools/extract/html-text.d.ts +27 -0
  258. package/dist/tools/extract/html-text.d.ts.map +1 -0
  259. package/dist/tools/extract/html-text.js +86 -0
  260. package/dist/tools/extract/html-text.js.map +1 -0
  261. package/dist/tools/extract/pdf-ocr.d.ts +58 -0
  262. package/dist/tools/extract/pdf-ocr.d.ts.map +1 -0
  263. package/dist/tools/extract/pdf-ocr.js +116 -0
  264. package/dist/tools/extract/pdf-ocr.js.map +1 -0
  265. package/dist/tools/extract/pdf.d.ts +65 -0
  266. package/dist/tools/extract/pdf.d.ts.map +1 -0
  267. package/dist/tools/extract/pdf.js +197 -0
  268. package/dist/tools/extract/pdf.js.map +1 -0
  269. package/dist/tools/extract/pptx.d.ts +32 -0
  270. package/dist/tools/extract/pptx.d.ts.map +1 -0
  271. package/dist/tools/extract/pptx.js +77 -0
  272. package/dist/tools/extract/pptx.js.map +1 -0
  273. package/dist/tools/extract/xlsx.d.ts +47 -0
  274. package/dist/tools/extract/xlsx.d.ts.map +1 -0
  275. package/dist/tools/extract/xlsx.js +111 -0
  276. package/dist/tools/extract/xlsx.js.map +1 -0
  277. package/dist/tools/finding-tool.d.ts +76 -0
  278. package/dist/tools/finding-tool.d.ts.map +1 -0
  279. package/dist/tools/finding-tool.js +125 -0
  280. package/dist/tools/finding-tool.js.map +1 -0
  281. package/dist/tools/messaging-tools.d.ts +41 -11
  282. package/dist/tools/messaging-tools.d.ts.map +1 -1
  283. package/dist/tools/messaging-tools.js +104 -55
  284. package/dist/tools/messaging-tools.js.map +1 -1
  285. package/dist/tools/neutts-synth.d.ts +27 -3
  286. package/dist/tools/neutts-synth.d.ts.map +1 -1
  287. package/dist/tools/neutts-synth.js +57 -13
  288. package/dist/tools/neutts-synth.js.map +1 -1
  289. package/dist/tools/pipeline-tool.d.ts +43 -1
  290. package/dist/tools/pipeline-tool.d.ts.map +1 -1
  291. package/dist/tools/pipeline-tool.js +13 -2
  292. package/dist/tools/pipeline-tool.js.map +1 -1
  293. package/dist/tools/read-extract.d.ts +116 -45
  294. package/dist/tools/read-extract.d.ts.map +1 -1
  295. package/dist/tools/read-extract.js +494 -158
  296. package/dist/tools/read-extract.js.map +1 -1
  297. package/dist/tools/registry.d.ts +2 -2
  298. package/dist/tools/registry.d.ts.map +1 -1
  299. package/dist/tools/registry.js +152 -23
  300. package/dist/tools/registry.js.map +1 -1
  301. package/dist/tools/subagent-refusal.d.ts +15 -0
  302. package/dist/tools/subagent-refusal.d.ts.map +1 -0
  303. package/dist/tools/subagent-refusal.js +18 -0
  304. package/dist/tools/subagent-refusal.js.map +1 -0
  305. package/dist/tools/subagent-spawner.d.ts +122 -0
  306. package/dist/tools/subagent-spawner.d.ts.map +1 -1
  307. package/dist/tools/subagent-spawner.js +249 -28
  308. package/dist/tools/subagent-spawner.js.map +1 -1
  309. package/dist/tools/tool-hooks.d.ts +177 -0
  310. package/dist/tools/tool-hooks.d.ts.map +1 -0
  311. package/dist/tools/tool-hooks.js +427 -0
  312. package/dist/tools/tool-hooks.js.map +1 -0
  313. package/dist/tools/tool-loop.d.ts +82 -0
  314. package/dist/tools/tool-loop.d.ts.map +1 -1
  315. package/dist/tools/tool-loop.js +167 -9
  316. package/dist/tools/tool-loop.js.map +1 -1
  317. package/dist/tools/tool-refusal.d.ts +68 -0
  318. package/dist/tools/tool-refusal.d.ts.map +1 -0
  319. package/dist/tools/tool-refusal.js +78 -0
  320. package/dist/tools/tool-refusal.js.map +1 -0
  321. package/dist/tools/toolsets.d.ts +8 -0
  322. package/dist/tools/toolsets.d.ts.map +1 -1
  323. package/dist/tools/toolsets.js +12 -2
  324. package/dist/tools/toolsets.js.map +1 -1
  325. package/dist/tools/vision-tools.d.ts +88 -83
  326. package/dist/tools/vision-tools.d.ts.map +1 -1
  327. package/dist/tools/vision-tools.js +134 -103
  328. package/dist/tools/vision-tools.js.map +1 -1
  329. package/dist/tools/worktree.d.ts +210 -0
  330. package/dist/tools/worktree.d.ts.map +1 -0
  331. package/dist/tools/worktree.js +374 -0
  332. package/dist/tools/worktree.js.map +1 -0
  333. package/dist/utils/format.d.ts +3 -0
  334. package/dist/utils/format.d.ts.map +1 -0
  335. package/dist/utils/format.js +32 -0
  336. package/dist/utils/format.js.map +1 -0
  337. package/dist/web-dashboard/attachment-extract.d.ts +64 -0
  338. package/dist/web-dashboard/attachment-extract.d.ts.map +1 -0
  339. package/dist/web-dashboard/attachment-extract.js +154 -0
  340. package/dist/web-dashboard/attachment-extract.js.map +1 -0
  341. package/dist/web-dashboard/chat-console.d.ts +108 -1
  342. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  343. package/dist/web-dashboard/chat-console.js +36 -0
  344. package/dist/web-dashboard/chat-console.js.map +1 -1
  345. package/dist/web-dashboard/hub-data.d.ts +37 -0
  346. package/dist/web-dashboard/hub-data.d.ts.map +1 -1
  347. package/dist/web-dashboard/hub-data.js +61 -1
  348. package/dist/web-dashboard/hub-data.js.map +1 -1
  349. package/dist/web-dashboard/process-env-inventory.d.ts +57 -0
  350. package/dist/web-dashboard/process-env-inventory.d.ts.map +1 -0
  351. package/dist/web-dashboard/process-env-inventory.js +96 -0
  352. package/dist/web-dashboard/process-env-inventory.js.map +1 -0
  353. package/dist/web-dashboard/server.d.ts +14 -0
  354. package/dist/web-dashboard/server.d.ts.map +1 -1
  355. package/dist/web-dashboard/server.js +493 -30
  356. package/dist/web-dashboard/server.js.map +1 -1
  357. package/dist/web-dashboard/src/types.d.ts +261 -48
  358. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  359. package/package.json +18 -3
  360. package/src/web-dashboard/public/assets/{index-Cyd6tIew.css → index-CIJ6FHHZ.css} +1 -1
  361. package/src/web-dashboard/public/assets/index-CsySzR41.js +207 -0
  362. package/src/web-dashboard/public/assets/index-CsySzR41.js.map +1 -0
  363. package/src/web-dashboard/public/index.html +2 -2
  364. package/dist/tools/child-agent-worker.js +0 -212
  365. package/src/web-dashboard/public/assets/index-CxDj7p6i.js +0 -207
  366. package/src/web-dashboard/public/assets/index-CxDj7p6i.js.map +0 -1
@@ -0,0 +1,1362 @@
1
+ /**
2
+ * WS0 (#22) — the drivers that actually run a turn on each surface.
3
+ *
4
+ * WHY THESE LIVE IN `src/` NOW. They used to live in `tests/parity/drivers.ts`
5
+ * and reached the stub through `vi.spyOn` on the real engine — which made the
6
+ * harness unrunnable outside the test runner, and left "prove every surface at
7
+ * par" as something only CI could do. The requirement is a `nuvira parity`
8
+ * command, so the drivers had to become a thing production code can call.
9
+ *
10
+ * The seam is CONFIGURATION, not a mock. Every surface resolves its provider
11
+ * through the SAME shared `resolveProvider` (`src/cli/router.ts`), so a temp
12
+ * `buffconfig.json` naming `groq` with `providers.groq.baseUrl` pointed at a
13
+ * loopback OpenAI-compatible stub makes the whole stack run for real: the real
14
+ * ChatCommand, the real console, the real gateway registry, the real execute
15
+ * command and the real forked child, each with a REAL provider object (the Groq
16
+ * adapter). Only the server on the other end of the socket is a stub — the
17
+ * definition of `transport` depth in `./scenarios.ts`, and the same depth the
18
+ * forked child was already driven at.
19
+ *
20
+ * NO TEST SEAM WAS ADDED TO PRODUCTION CODE. The surfaces already accept what is
21
+ * needed: `answerOnce`/`ChatConsole.answer` take `provider`/`model`, the gateway
22
+ * derives the pair from its own config (`registry.ts:1797`), and the child reads
23
+ * its own `buffconfig.json`. Nothing in `chat.ts`, `chat-console.ts`,
24
+ * `gateway/registry.ts`, `execute.ts` or the spawner changed to make this work.
25
+ *
26
+ * ISOLATION IS THE POINT, AND IT IS PROCESS-LOCAL. A run points
27
+ * `NUVIRA_CONFIG_DIR` and `NUVIRA_MEMORY_DIR` at a throwaway directory for its
28
+ * whole life, so the stub never touches the developer's real profile — no real
29
+ * API key, no real response cache, no real model registry, no real gateway log.
30
+ * That is the same convention `src/config/paths.ts` documents and the same one
31
+ * the test drivers relied on. It cannot leak into a separately-running dashboard
32
+ * or gateway: a `nuvira parity` invocation is its own process.
33
+ *
34
+ * EVERY SURFACE'S OBSERVATION IS READ FROM THAT SURFACE'S OWN REPORT — the
35
+ * engine's return for the CLI, the console's result for the dashboard, the
36
+ * gateway's `inbound.chat` log record for the gateway, the command's result for
37
+ * execute, the child's own progress frames for the subagent. Nothing is
38
+ * reconstructed here, so a surface that stops reporting its status, attribution
39
+ * or tool calls goes red instead of quietly losing the fact.
40
+ */
41
+ import { createServer } from 'node:http';
42
+ import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
43
+ import { tmpdir } from 'node:os';
44
+ import { join } from 'node:path';
45
+ import { gunzipSync } from 'node:zlib';
46
+ import { getCache } from '../context/cache.js';
47
+ import { readGatewayLog } from '../gateway/gateway-log.js';
48
+ import { debugLogDir, readLatestDebugLog } from '../observability/debug-log.js';
49
+ // WS3 (#25) — the far end of the export the surfaces run, and the two names that
50
+ // define the portable span tree (`src/observability/otel.ts`).
51
+ import { otelEnableVarName, shutdownSpans, TOOL_SPAN_PREFIX, TURN_SPAN_NAME, } from '../observability/otel.js';
52
+ // WS4 (#26) — the hook ENV names the harness declares through, and the phases
53
+ // they exist for. Imported from the module that defines them so a renamed
54
+ // variable cannot leave the harness silently declaring nothing (which would read
55
+ // as "no surface fired a hook", the failure this row is meant to catch).
56
+ import { TOOL_HOOK_ENV, TOOL_HOOK_PHASES } from '../tools/tool-hooks.js';
57
+ import { noDebugLog, noFault, noIsolation, noOtelExport, noResume, noToolHooks, turnStatus, } from './observation.js';
58
+ // WS6 (#28) — the fault protocol. The harness OWNS the provider faults (the stub
59
+ // answers them) and DECLARES the seam faults (`tool`/`ipc`), so both halves of the
60
+ // workstream are driven through the same run rather than two harnesses.
61
+ import { FAULT_ENV, faultMessage, formatFaultPlan, resetFaultInjector } from '../runtime/fault-injection.js';
62
+ // WS5 (#27) — the two env keys the harness declares these capabilities THROUGH,
63
+ // imported from the modules that define them so a renamed variable cannot leave
64
+ // the harness silently declaring nothing (which would read as "no surface
65
+ // isolated its turn", the failure this row is meant to catch).
66
+ import { WORKTREE_ENABLE_ENV } from '../tools/worktree.js';
67
+ import { RESUME_ENABLE_ENV } from '../learning/step-checkpoint.js';
68
+ /**
69
+ * Where every driver stubs. Transport depth is not a compromise here — it is
70
+ * what makes the comparison honest: the provider OBJECT is the real adapter and
71
+ * the turn code above it is untouched. The runner folds `provider` and
72
+ * `transport` into one comparable class (`./scenarios.ts`, rule 1), so these
73
+ * observations compare with anything else that runs the real turn code.
74
+ */
75
+ export const DRIVER_DEPTH = 'transport';
76
+ /** The provider id every surface is configured with, so the comparison is at-par. */
77
+ export const PARITY_PROVIDER_TYPE = 'groq';
78
+ /** The one model the stub serves. The surfaces are pinned to it, so all five agree. */
79
+ export const PARITY_MODEL = 'parity-stub-model';
80
+ /** The stub's API key. Never a real credential — the stub never validates it. */
81
+ const PARITY_API_KEY = 'parity-stub-key';
82
+ /**
83
+ * The surfaces this module can drive. Declared as data so a caller can assert
84
+ * the driver list covers the registry without paying for a live harness — a
85
+ * surface missing here would otherwise be counted as neither covered nor
86
+ * blocked, the silent hole this harness exists to prevent.
87
+ */
88
+ export const PARITY_DRIVER_SURFACES = [
89
+ 'cli-chat',
90
+ 'dashboard-chat',
91
+ 'gateway-chat',
92
+ 'cli-execute',
93
+ 'subagent',
94
+ ];
95
+ async function startStub(scenario) {
96
+ let chatCalls = 0;
97
+ let faultsServed = 0;
98
+ // Only a PROVIDER-site fault is the stub's to serve. A `tool`/`ipc` fault is
99
+ // declared to the running agent instead (`withTurnEnvelope`), which is what
100
+ // makes the seam, rather than this stub, the thing under test there.
101
+ const providerFault = scenario.fault?.site === 'provider' ? scenario.fault : null;
102
+ const server = createServer((req, res) => {
103
+ const chunks = [];
104
+ req.on('data', (chunk) => chunks.push(chunk));
105
+ req.on('end', () => {
106
+ const json = (body) => {
107
+ res.writeHead(200, { 'content-type': 'application/json' });
108
+ res.end(JSON.stringify(body));
109
+ };
110
+ if (req.url?.endsWith('/models')) {
111
+ // A reachability probe or a model-list validation against an
112
+ // OpenAI-compatible endpoint asks here. The stub serves the ONE model
113
+ // the surfaces are pinned to, so `resolveWorkingModel` keeps the pin
114
+ // instead of repairing it to some other provider's default.
115
+ // DELIBERATELY NOT FAULTED: a faulted probe would make the surfaces fail
116
+ // in their ROUTING rather than in their turn, which is a different row
117
+ // (and would let a surface pass without ever attempting the call).
118
+ return json({ data: [{ id: PARITY_MODEL, object: 'model' }] });
119
+ }
120
+ if (req.url?.endsWith('/chat/completions')) {
121
+ chatCalls += 1;
122
+ // WS6 (#28) — the DECLARED provider fault, served on the wire so the REAL
123
+ // adapter's error mapping is what runs. `faultMessage` is shared with the
124
+ // seam, so a fault reads the same words wherever it came from, and a body
125
+ // that cannot be parsed is a distinct kind rather than a second flavour of
126
+ // "error" — a response that arrives but says nothing is its own failure.
127
+ if (providerFault && faultsServed < providerFault.times) {
128
+ faultsServed += 1;
129
+ if (providerFault.kind === 'malformed') {
130
+ res.writeHead(200, { 'content-type': 'application/json' });
131
+ res.end('{"choices": [{"message": {"content": '); // truncated on purpose
132
+ return;
133
+ }
134
+ res.writeHead(providerFault.kind === 'unavailable' ? 503 : 500, {
135
+ 'content-type': 'application/json',
136
+ });
137
+ res.end(JSON.stringify({
138
+ error: { message: faultMessage(providerFault, 'the model call') },
139
+ }));
140
+ return;
141
+ }
142
+ let body = {};
143
+ try {
144
+ body = JSON.parse(Buffer.concat(chunks).toString('utf8') || '{}');
145
+ }
146
+ catch {
147
+ body = {};
148
+ }
149
+ const toolAlreadyRan = (body.messages ?? []).some((m) => m?.role === 'tool');
150
+ const wantTool = Boolean(scenario.toolCall) && !toolAlreadyRan;
151
+ const content = wantTool ? '' : scenario.answer;
152
+ const toolCalls = wantTool
153
+ ? [
154
+ {
155
+ id: 'parity_call_1',
156
+ type: 'function',
157
+ function: {
158
+ name: scenario.toolCall.tool,
159
+ arguments: JSON.stringify(scenario.toolCall.args),
160
+ },
161
+ },
162
+ ]
163
+ : undefined;
164
+ // The dashboard console passes `onToken`, so the real Groq adapter takes
165
+ // the STREAMING path (`generateToolsStream` -> SSE). Answering that with
166
+ // a JSON body reads as an empty response and the loop retries until its
167
+ // budget — measured, and the reason this branch exists. Both shapes are
168
+ // served so every surface can be driven the way it really talks.
169
+ if (body.stream) {
170
+ res.writeHead(200, { 'content-type': 'text/event-stream', 'cache-control': 'no-cache' });
171
+ const chunk = (delta, finish) => `data: ${JSON.stringify({
172
+ choices: [{ delta, ...(finish ? { finish_reason: finish } : {}) }],
173
+ })}\n\n`;
174
+ const fragments = [];
175
+ if (wantTool) {
176
+ fragments.push(chunk({
177
+ role: 'assistant',
178
+ content: '',
179
+ tool_calls: toolCalls.map((call, index) => ({ index, ...call })),
180
+ }));
181
+ }
182
+ else {
183
+ fragments.push(chunk({ role: 'assistant', content }));
184
+ }
185
+ fragments.push(chunk({}, wantTool ? 'tool_calls' : 'stop'));
186
+ fragments.push('data: [DONE]\n\n');
187
+ res.end(fragments.join(''));
188
+ return;
189
+ }
190
+ return json({
191
+ choices: [
192
+ {
193
+ message: {
194
+ content,
195
+ ...(toolCalls ? { tool_calls: toolCalls } : {}),
196
+ },
197
+ },
198
+ ],
199
+ });
200
+ }
201
+ res.writeHead(404, { 'content-type': 'application/json' });
202
+ res.end('{}');
203
+ });
204
+ });
205
+ await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve));
206
+ const { port } = server.address();
207
+ return {
208
+ baseUrl: `http://127.0.0.1:${port}/v1`,
209
+ chatCalls: () => chatCalls,
210
+ faultsServed: () => faultsServed,
211
+ close: () => new Promise((resolve) => server.close(() => resolve())),
212
+ };
213
+ }
214
+ /** Pull the comparable facts out of one OTLP JSON payload, ignoring the rest. */
215
+ function collectSpansInto(payload, into, onServiceName) {
216
+ const resourceSpans = payload?.resourceSpans;
217
+ if (!Array.isArray(resourceSpans))
218
+ return;
219
+ for (const entry of resourceSpans) {
220
+ const attributes = entry?.resource?.attributes;
221
+ if (Array.isArray(attributes)) {
222
+ for (const attribute of attributes) {
223
+ const a = attribute;
224
+ if (a?.key === 'service.name' && typeof a.value?.stringValue === 'string') {
225
+ onServiceName(a.value.stringValue);
226
+ }
227
+ }
228
+ }
229
+ const scopeSpans = entry?.scopeSpans;
230
+ if (!Array.isArray(scopeSpans))
231
+ continue;
232
+ for (const scope of scopeSpans) {
233
+ const spans = scope?.spans;
234
+ if (!Array.isArray(spans))
235
+ continue;
236
+ for (const span of spans) {
237
+ const s = span;
238
+ if (typeof s?.name !== 'string')
239
+ continue;
240
+ into.push({
241
+ name: s.name,
242
+ traceId: typeof s.traceId === 'string' ? s.traceId : '',
243
+ spanId: typeof s.spanId === 'string' ? s.spanId : '',
244
+ parentSpanId: typeof s.parentSpanId === 'string' ? s.parentSpanId : '',
245
+ });
246
+ }
247
+ }
248
+ }
249
+ }
250
+ async function startOtlpCollector() {
251
+ const received = [];
252
+ let service = null;
253
+ let requests = 0;
254
+ const server = createServer((req, res) => {
255
+ const chunks = [];
256
+ req.on('data', (chunk) => chunks.push(chunk));
257
+ req.on('end', () => {
258
+ requests += 1;
259
+ try {
260
+ const raw = Buffer.concat(chunks);
261
+ // The SDK gzips when it is told to; decoding it here means a run that
262
+ // enables compression is MEASURED rather than silently read as empty.
263
+ const body = req.headers['content-encoding'] === 'gzip' ? gunzipSync(raw) : raw;
264
+ collectSpansInto(JSON.parse(body.toString('utf8')), received, (name) => {
265
+ service ??= name;
266
+ });
267
+ }
268
+ catch {
269
+ // A body we cannot parse is a span we report as MISSING — never a crash
270
+ // in the harness, and never a silent pass.
271
+ }
272
+ res.writeHead(200, { 'content-type': 'application/json' });
273
+ res.end('{}');
274
+ });
275
+ });
276
+ await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve));
277
+ const { port } = server.address();
278
+ return {
279
+ endpoint: `http://127.0.0.1:${port}/v1/traces`,
280
+ spans: () => [...received],
281
+ serviceName: () => service,
282
+ requests: () => requests,
283
+ close: () => new Promise((resolve) => {
284
+ // The exporter holds keep-alive sockets and `close()` alone waits for
285
+ // them. Tearing them down explicitly is what stops a harness run from
286
+ // hanging on its own collector after the last span arrived.
287
+ server.close(() => resolve());
288
+ server.closeAllConnections?.();
289
+ }),
290
+ };
291
+ }
292
+ /**
293
+ * Reduce the spans a collector received to the compared projection.
294
+ *
295
+ * The collector hands spans over in COMPLETION order (measured), so the whole
296
+ * reduction is order-insensitive: sorted names, sorted edges. A `parentSpanId`
297
+ * matching no span in this collector is a REMOTE parent (a child process that
298
+ * continued a trace begun elsewhere) — it contributes no edge, because an edge to
299
+ * a name we never received cannot compare, and it is recorded on `remoteParent`
300
+ * for the scenario's own assertion instead.
301
+ */
302
+ function otelObsOf(collector) {
303
+ const received = collector.spans();
304
+ if (received.length === 0)
305
+ return noOtelExport();
306
+ const nameById = new Map(received.filter((s) => s.spanId).map((s) => [s.spanId, s.name]));
307
+ const names = received.map((s) => s.name).sort();
308
+ const edges = new Set();
309
+ for (const span of received) {
310
+ if (!span.parentSpanId)
311
+ continue;
312
+ const parent = nameById.get(span.parentSpanId);
313
+ if (parent)
314
+ edges.add(`${parent} → ${span.name}`);
315
+ }
316
+ const traceIds = [...new Set(received.map((s) => s.traceId).filter(Boolean))];
317
+ const turn = received.find((s) => s.name === TURN_SPAN_NAME) ?? null;
318
+ return {
319
+ exported: true,
320
+ spans: names,
321
+ edges: [...edges].sort(),
322
+ turnSpans: received.filter((s) => s.name === TURN_SPAN_NAME).length,
323
+ toolSpans: names.filter((name) => name.startsWith(TOOL_SPAN_PREFIX)),
324
+ singleTrace: traceIds.length === 1,
325
+ serviceName: collector.serviceName(),
326
+ traceId: traceIds.length === 1 ? traceIds[0] : (turn?.traceId ?? null),
327
+ remoteParent: turn && turn.parentSpanId && !nameById.has(turn.parentSpanId) ? turn.parentSpanId : null,
328
+ };
329
+ }
330
+ /**
331
+ * Run one surface's turn with a fresh collector in front of the export path.
332
+ *
333
+ * Returns the turn's own value AND the projection read from the collector after
334
+ * it finished, so a driver wraps exactly the call that talks to the model and
335
+ * nothing else. The endpoint is set for the duration of that call and restored
336
+ * after, because the SDK reads it when it BUILDS the provider — which is why the
337
+ * harness resets the provider before each surface (`shutdownSpans` in the driver
338
+ * wrapper) instead of trusting one endpoint to serve them all.
339
+ */
340
+ async function withOtlpCollector(body) {
341
+ const collector = await startOtlpCollector();
342
+ const previousEndpoint = process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT;
343
+ process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = collector.endpoint;
344
+ try {
345
+ const value = await body();
346
+ return { value, otel: otelObsOf(collector) };
347
+ }
348
+ finally {
349
+ if (previousEndpoint === undefined)
350
+ delete process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT;
351
+ else
352
+ process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = previousEndpoint;
353
+ await collector.close();
354
+ }
355
+ }
356
+ // ─── The operator's tool hooks ──────────────────────────────────────────────
357
+ /** Where a declared hook appends the invocations it received (harness-owned). */
358
+ export const TOOL_HOOK_LOG_VAR = 'NUVIRA_TOOL_HOOK_LOG';
359
+ /** Which tools a declared `before` hook vetoes (harness-owned). */
360
+ export const TOOL_HOOK_DENY_VAR = 'NUVIRA_TOOL_HOOK_DENY';
361
+ /**
362
+ * WS4 (#26) — the hook script the harness DECLARES, exactly as written.
363
+ *
364
+ * It is a real operator hook: a command that reads the call as JSON on stdin and
365
+ * may veto it on stdout. It reads two variables the harness sets — where to
366
+ * append the invocation it received, and which tools this scenario's `before`
367
+ * hook denies — so ONE script serves every scenario and every phase, and the log
368
+ * it writes is the record this row is compared on.
369
+ *
370
+ * Why a script at all, rather than a spy on the hook seam: the claim is "an
371
+ * operator's declared command runs", and only a real process proves the path an
372
+ * operator would actually take — the spawn, the stdin pipe, the JSON contract, the
373
+ * exit code. A spy would only prove the surface called its own helper.
374
+ */
375
+ const TOOL_HOOK_SCRIPT = `#!/usr/bin/env node
376
+ // WS4 (#26) — the operator hook the parity harness declares. See
377
+ // src/parity/drivers.ts for the contract and why it is a real process.
378
+ import { appendFileSync } from 'node:fs';
379
+
380
+ let raw = '';
381
+ process.stdin.setEncoding('utf8');
382
+ for await (const chunk of process.stdin) raw += chunk;
383
+
384
+ let payload = {};
385
+ try {
386
+ payload = JSON.parse(raw);
387
+ } catch {
388
+ // Not JSON means this was never handed a real payload: exit non-zero so the
389
+ // surface reports a problem instead of reading silence as a decision.
390
+ process.stderr.write('parity hook: stdin was not the documented JSON payload');
391
+ process.exit(2);
392
+ }
393
+
394
+ const denyList = (process.env.${TOOL_HOOK_DENY_VAR} ?? '')
395
+ .split(',')
396
+ .map((name) => name.trim())
397
+ .filter(Boolean);
398
+ const denied = payload.phase === 'before' && denyList.includes(payload.tool);
399
+
400
+ const log = process.env.${TOOL_HOOK_LOG_VAR};
401
+ if (log) {
402
+ appendFileSync(
403
+ log,
404
+ JSON.stringify({
405
+ phase: payload.phase,
406
+ tool: payload.tool,
407
+ surface: payload.surface ?? null,
408
+ ok: typeof payload.ok === 'boolean' ? payload.ok : null,
409
+ decision: denied ? 'deny' : null,
410
+ hook: payload.hook,
411
+ }) + '\\n',
412
+ );
413
+ }
414
+
415
+ if (denied) {
416
+ process.stdout.write(
417
+ JSON.stringify({ decision: 'deny', reason: 'parity: ' + payload.tool + ' is not allowed to run' }),
418
+ );
419
+ }
420
+ `;
421
+ /**
422
+ * Reduce the hook's log AND the surface's own tool lifecycle to the projection.
423
+ *
424
+ * Two witnesses, deliberately: the log says what the hook was asked and how it
425
+ * answered, and `observation.toolCalls` says what the SURFACE did with that
426
+ * answer. `vetoReported` needs both to agree; `vetoLeaked` needs only the surface
427
+ * to show a denied tool succeeding. See `ToolHooksObs` for why neither side alone
428
+ * is trusted.
429
+ */
430
+ function toolHooksObsOf(logPath, observation) {
431
+ let lines;
432
+ try {
433
+ lines = readFileSync(logPath, 'utf8').split('\n').filter((line) => line.trim() !== '');
434
+ }
435
+ catch {
436
+ // No log means the hook never ran (or never wrote) — the honest value, which
437
+ // a scenario that declared hooks reads as a failure rather than a neutral.
438
+ return noToolHooks();
439
+ }
440
+ const invocations = new Set();
441
+ const denied = new Set();
442
+ const surfacesSeen = new Set();
443
+ for (const line of lines) {
444
+ let entry;
445
+ try {
446
+ entry = JSON.parse(line);
447
+ }
448
+ catch {
449
+ continue;
450
+ }
451
+ const phase = typeof entry.phase === 'string' ? entry.phase : 'unknown';
452
+ const tool = typeof entry.tool === 'string' ? entry.tool : 'unknown';
453
+ invocations.add(`${phase}:${tool}`);
454
+ if (typeof entry.surface === 'string' && entry.surface !== '')
455
+ surfacesSeen.add(entry.surface);
456
+ if (phase === 'before' && entry.decision === 'deny')
457
+ denied.add(tool);
458
+ }
459
+ const deniedTools = [...denied].sort();
460
+ // The surface's OWN outcomes for the denied tools: `true` = reported success
461
+ // (so the call ran — a leak), `false` = reported as a failed call.
462
+ const outcomes = deniedTools.map((tool) => observation.toolCalls.filter((call) => call.tool === tool).map((call) => call.ok === true));
463
+ return {
464
+ invocations: [...invocations].sort(),
465
+ denied: deniedTools,
466
+ vetoReported: deniedTools.length > 0 && outcomes.every((perTool) => perTool.some((ok) => ok === false)),
467
+ vetoLeaked: outcomes.some((perTool) => perTool.some((ok) => ok === true)),
468
+ surfacesSeen: [...surfacesSeen].sort(),
469
+ };
470
+ }
471
+ /**
472
+ * Run one surface's turn with this scenario's hooks DECLARED, and read back what
473
+ * the hook received.
474
+ *
475
+ * The declarations are environment variables for the duration of the call and are
476
+ * restored afterwards, for the same reason the OTLP endpoint is: the surfaces
477
+ * read them at call time, so a scenario that declared a hook must not leave it
478
+ * declared for the next one — a veto that leaked into the next scenario would
479
+ * look like that scenario's own behaviour.
480
+ *
481
+ * The log file is per SURFACE as well as per scenario, so the child's invocations
482
+ * (written from its own process, which inherits the path) cannot be attributed to
483
+ * an in-process surface that ran before it.
484
+ */
485
+ async function withToolHooks(ws, surface, scenario, body) {
486
+ const declared = scenario.hooks;
487
+ const logPath = join(ws.root, `tool-hook-${scenario.id}-${surface}.jsonl`);
488
+ rmSync(logPath, { force: true });
489
+ const previous = new Map();
490
+ const set = (name, value) => {
491
+ previous.set(name, process.env[name]);
492
+ if (value === undefined)
493
+ delete process.env[name];
494
+ else
495
+ process.env[name] = value;
496
+ };
497
+ const deny = declared?.deny?.filter((tool) => tool.trim() !== '') ?? [];
498
+ for (const phase of TOOL_HOOK_PHASES) {
499
+ set(TOOL_HOOK_ENV[phase], declared?.phases.includes(phase) ? `node ${ws.hookScript}` : undefined);
500
+ }
501
+ set(TOOL_HOOK_LOG_VAR, declared ? logPath : undefined);
502
+ set(TOOL_HOOK_DENY_VAR, deny.length > 0 ? deny.join(',') : undefined);
503
+ try {
504
+ const observation = await body();
505
+ return { ...observation, hooks: toolHooksObsOf(logPath, observation) };
506
+ }
507
+ finally {
508
+ for (const [name, value] of previous) {
509
+ if (value === undefined)
510
+ delete process.env[name];
511
+ else
512
+ process.env[name] = value;
513
+ }
514
+ }
515
+ }
516
+ // ─── The turn envelope (WS5: isolation and resume) ──────────────────────────
517
+ /**
518
+ * Run one surface's turn with this scenario's isolation and resume DECLARED, and
519
+ * reduce the surface's own reports to what is compared.
520
+ *
521
+ * THE RESUME PROBE IS A PAIR OF TURNS, and it has to be: "a resumed run reuses
522
+ * unchanged steps instead of re-paying for every model call" is a statement about
523
+ * two runs — one that writes the record and one that reads it — so a single turn
524
+ * cannot produce the fact. The FIRST turn is the one everything else about the
525
+ * scenario is read from (a fully replayed turn reaches no model at all, and the
526
+ * runner refuses to compare such a turn); the SECOND contributes only its resume
527
+ * fields.
528
+ *
529
+ * THE RESPONSE CACHE IS CLEARED BEFORE EACH TURN, and skipping that would make
530
+ * this row lie in the most convenient direction: the second turn sends the same
531
+ * message as the first, so a cache hit would answer it without reaching the loop
532
+ * at all — no record read, no step replayed, and a `resuming: false` that a
533
+ * harness comparing only model counts would read as agreement. The driver's own
534
+ * turns clear it too; this is the second, independent guard.
535
+ *
536
+ * Both declarations are the ENVIRONMENT (`NUVIRA_ISOLATE` / `NUVIRA_RESUME`) and
537
+ * are restored afterwards, exactly like the OTLP endpoint and the hooks: the
538
+ * surfaces read them at call time, so a declaration that leaked into the next
539
+ * scenario would look like that scenario's own behaviour. The environment is also
540
+ * what makes ONE declaration cover all five surfaces — the dashboard server, the
541
+ * gateway and the forked child have no flags to carry.
542
+ */
543
+ async function withTurnEnvelope(scenario, surface, body) {
544
+ const askedIsolation = scenario.isolation === true;
545
+ const askedResume = scenario.resume === true;
546
+ // WS6 (#28) — a `tool`/`ipc` fault is DECLARED to the running agent (the seam),
547
+ // while a `provider` fault is served by the stub (so the real adapter's error
548
+ // mapping runs and the model call still happens). Both are set explicitly,
549
+ // including the `undefined` case: a declaration inherited from the developer's
550
+ // shell would make the scenarios that DO NOT ask for a fault inject one anyway.
551
+ const seamFault = scenario.fault && scenario.fault.site !== 'provider' ? scenario.fault : null;
552
+ const previousIsolation = process.env[WORKTREE_ENABLE_ENV];
553
+ const previousResume = process.env[RESUME_ENABLE_ENV];
554
+ const previousFault = process.env[FAULT_ENV];
555
+ const set = (name, value) => {
556
+ if (value === undefined)
557
+ delete process.env[name];
558
+ else
559
+ process.env[name] = value;
560
+ };
561
+ /**
562
+ * The record this surface's probe uses — NAMED, and namespaced by surface.
563
+ *
564
+ * The auto id is `checkpointIdFor(goal, cwd)`, which is the right default for a
565
+ * human (`--resume` means "the last run of this ask, here") and exactly wrong for
566
+ * this harness: every in-process surface runs the SAME ask in the SAME directory,
567
+ * so they would all read ONE record and a surface could "replay" a tree another
568
+ * surface recorded — measured, and it made two surfaces pass for a reason that had
569
+ * nothing to do with them. A per-surface id keeps each probe's record its own,
570
+ * which is also the path an operator uses to resume a named run.
571
+ */
572
+ const resumeId = `parity-${scenario.id}-${surface}`;
573
+ try {
574
+ set(WORKTREE_ENABLE_ENV, askedIsolation ? '1' : undefined);
575
+ // BOTH turns are asked to resume, and that is not a formality: the FIRST one is
576
+ // what WRITES the record the second replays. A probe whose first turn ran
577
+ // without a ledger would compare a resumed turn against an empty record —
578
+ // replayed 0, model calls unchanged — and report that as the capability working.
579
+ set(RESUME_ENABLE_ENV, askedResume ? resumeId : undefined);
580
+ set(FAULT_ENV, seamFault ? formatFaultPlan(seamFault) : undefined);
581
+ // The injector is cached per declaration, and this declaration is fresh for
582
+ // this surface — so it starts with a full allowance either way.
583
+ resetFaultInjector();
584
+ await clearResponseCache();
585
+ const first = await body();
586
+ if (!askedResume)
587
+ return first;
588
+ await clearResponseCache();
589
+ const second = await body();
590
+ // Only the resume fields come from the second turn: everything else about the
591
+ // scenario is read from the first one, which is the turn that reached a model.
592
+ return { ...first, resume: second.resume };
593
+ }
594
+ finally {
595
+ set(WORKTREE_ENABLE_ENV, previousIsolation);
596
+ set(RESUME_ENABLE_ENV, previousResume);
597
+ set(FAULT_ENV, previousFault);
598
+ resetFaultInjector();
599
+ }
600
+ }
601
+ /**
602
+ * The response cache is SHARED, ON DISK, AND ON BY DEFAULT for `answerOnce`
603
+ * (`src/cli/chat.ts:690`), keyed by `provider:model:prompt`. Every surface of a
604
+ * run sends the SAME scenario message, so without clearing between surfaces the
605
+ * second one would be served from the first one's entry — same answer, no model
606
+ * call, no tool lifecycle, and a flawless "agreement" between two replays.
607
+ *
608
+ * Isolating `NUVIRA_MEMORY_DIR` already makes the cache file fresh per run, but
609
+ * within a run the entries still collide, so this clears before each surface.
610
+ * The runner's `unreached-model` refusal (`./scenarios.ts`, rule 4) is the
611
+ * second, independent guard: a zero from a replay is refused, not compared.
612
+ */
613
+ async function clearResponseCache() {
614
+ try {
615
+ await getCache().clear();
616
+ }
617
+ catch {
618
+ // The cache may never break a turn — and the `modelCalls` guard catches a
619
+ // stale entry anyway.
620
+ }
621
+ }
622
+ function createWorkspace() {
623
+ const root = mkdtempSync(join(tmpdir(), 'buff-parity-'));
624
+ const configDir = join(root, 'config');
625
+ const memoryDir = join(root, 'memory');
626
+ mkdirSync(configDir, { recursive: true });
627
+ mkdirSync(memoryDir, { recursive: true });
628
+ // WS4 — the hook command the harness declares. Written once per run, so every
629
+ // scenario's hooks are the SAME program and a difference between scenarios can
630
+ // only come from the declaration, not from the script.
631
+ const hookScript = join(root, 'tool-hook.mjs');
632
+ writeFileSync(hookScript, TOOL_HOOK_SCRIPT);
633
+ const workspace = {
634
+ root,
635
+ configDir,
636
+ memoryDir,
637
+ hookScript,
638
+ useStub(baseUrl) {
639
+ // `buffconfig.json` is the file `ConfigManager` reads (`config/manager.ts`
640
+ // -> `config/paths.ts`). The provider object the surfaces build from it is
641
+ // the REAL Groq adapter; `baseUrl` is the override it honours.
642
+ writeFileSync(join(configDir, 'buffconfig.json'), JSON.stringify({
643
+ defaultProvider: PARITY_PROVIDER_TYPE,
644
+ providers: {
645
+ [PARITY_PROVIDER_TYPE]: { apiKey: PARITY_API_KEY, model: PARITY_MODEL, baseUrl },
646
+ },
647
+ }, null, 2));
648
+ },
649
+ };
650
+ return workspace;
651
+ }
652
+ /**
653
+ * Reduce a surface's own isolation report to the compared projection.
654
+ *
655
+ * Takes the shape both the in-process surfaces and the subagent manager report
656
+ * (a base commit, a file list and whether the directory was removed) rather than
657
+ * one of their concrete types, because the two are produced by different code on
658
+ * purpose: the in-process turn makes its own worktree, the parent makes the
659
+ * child's. What must agree is the RESULT, and that is what this reduces.
660
+ */
661
+ function isolationObsOf(asked, report) {
662
+ if (!report)
663
+ return { ...noIsolation(), asked };
664
+ return {
665
+ asked,
666
+ isolated: true,
667
+ // Sorted, so the comparison is over the SET of files the run changed: a diff
668
+ // is measured from `git diff`, whose order is the repository's, not the
669
+ // surface's.
670
+ files: [...report.diff.files].sort(),
671
+ removed: report.removed,
672
+ base: report.base,
673
+ };
674
+ }
675
+ /**
676
+ * Reduce a surface's own resume report to the compared projection.
677
+ *
678
+ * `asked` with no report is the honest description of a surface that ignored the
679
+ * request: `resuming: false` against every other surface's `true`, which
680
+ * `compare` reports as a difference rather than as agreement about nothing.
681
+ */
682
+ function resumeObsOf(asked, report) {
683
+ if (!report)
684
+ return { ...noResume(), asked };
685
+ return {
686
+ asked,
687
+ resuming: true,
688
+ replayed: report.replayed,
689
+ modelCalls: report.modelCalls,
690
+ saved: report.saved,
691
+ };
692
+ }
693
+ /**
694
+ * Read a surface's OWN session debug log and reduce its header to the compared
695
+ * facts — WS2.
696
+ *
697
+ * Read from DISK rather than from a return value, for the same reason the
698
+ * gateway driver reads `inbound.chat` from the gateway log: the capability is
699
+ * "a file you can attach to a bug report", so the artifact itself is the
700
+ * evidence. `written: false` when the surface produced nothing, which the
701
+ * harness refuses to read as agreement (the run turns logging on for every
702
+ * surface).
703
+ */
704
+ function debugLogOf(surface, dir = debugLogDir()) {
705
+ const found = readLatestDebugLog(surface, dir);
706
+ if (!found)
707
+ return noDebugLog();
708
+ return {
709
+ written: true,
710
+ provider: found.header.provider,
711
+ model: found.header.model,
712
+ transport: found.header.transport,
713
+ };
714
+ }
715
+ /**
716
+ * Reduce a surface's own answer to the shared observation.
717
+ *
718
+ * MEASURED, and found by this harness on its first real run: the surfaces do NOT
719
+ * share a result contract. `ChatCommand.answerOnce` reports `generationFailed`
720
+ * and no `ok`; the console SYNTHESISES `ok`. Mapping each surface's OWN contract
721
+ * rather than inventing a common one keeps that seam visible instead of hiding
722
+ * it — which is what a parity harness is for.
723
+ */
724
+ /**
725
+ * WS6 (#28) — whether this surface's OWN turn shows the fault it was given.
726
+ *
727
+ * Derived rather than counted, because the counter cannot cross the fork (see
728
+ * `FaultObs`). The rule is the fault's own contract:
729
+ *
730
+ * - `tool` — the named call (or any call, when none is named) is reported
731
+ * FAILED. A surface that swallowed the injected `Error:` and
732
+ * reported the call as ok reads as `took: false`.
733
+ * - `provider` / `ipc` — the turn did not COMPLETE. A surface that produced an
734
+ * answer anyway reads as `took: false`, which is precisely the
735
+ * false-success shape this workstream exists to catch.
736
+ */
737
+ function faultObsOf(scenario, toolCalls, status) {
738
+ const plan = scenario.fault;
739
+ if (!plan)
740
+ return noFault();
741
+ const took = plan.site === 'tool'
742
+ ? toolCalls.some((call) => (plan.match === undefined || call.tool === plan.match) && call.ok === false)
743
+ : status !== 'completed';
744
+ return { asked: true, site: plan.site, kind: plan.kind, took };
745
+ }
746
+ function toObservation(surface, scenario, answer, toolCalls, modelCalls) {
747
+ // WS6 (#28) — `generationFailed` outranks `ok`, through the ONE helper every
748
+ // surface's read goes through (see `turnStatus`). A served-but-failed turn is a
749
+ // failure; reading the request-level `ok` first is what let the provider-fault
750
+ // row catch two surfaces calling an empty generation a completion.
751
+ const status = turnStatus(answer);
752
+ const succeeded = status === 'completed';
753
+ return {
754
+ surface,
755
+ engine: 'loop',
756
+ status,
757
+ // Recorded, never inferred: the runner refuses a zero (rule 4 in
758
+ // ./scenarios.ts) instead of reading a cache replay as agreement.
759
+ modelCalls,
760
+ ...(answer.provider ? { provider: answer.provider } : {}),
761
+ ...(answer.model ? { model: answer.model } : {}),
762
+ ...(answer.transport ? { transport: answer.transport } : {}),
763
+ toolCalls,
764
+ // WS1 — recorded findings, in order. `[]` when the surface reported none.
765
+ findings: answer.findings ?? [],
766
+ // WS2 — the session debug log's header. `written: false` when the surface
767
+ // produced none, which the harness (logging ON) reads as a failure.
768
+ debugLog: answer.debugLog ?? noDebugLog(),
769
+ // WS3 — the span tree the collector received. `exported: false` when nothing
770
+ // arrived, which the harness (export ON) reads as a failure.
771
+ otel: answer.otel ?? noOtelExport(),
772
+ // WS4 — replaced by the driver wrapper with what the operator's declared hook
773
+ // actually received (`withToolHooks`): the log is written by the HOOK, and
774
+ // only the wrapper knows which file this surface's run was pointed at.
775
+ hooks: noToolHooks(),
776
+ // WS5 — the surface's OWN report of the isolation it ran with, and of what its
777
+ // resume replayed. `asked` comes from the scenario, which is what makes
778
+ // "asked for it and did not do it" a difference rather than a tautology. The
779
+ // resume field is REPLACED by `withTurnEnvelope` for the resumed scenario,
780
+ // whose second turn is the one that can answer it.
781
+ isolation: isolationObsOf(scenario.isolation === true, answer.worktree),
782
+ resume: resumeObsOf(scenario.resume === true, answer.resume),
783
+ // WS6 (#28) — the declared fault, and whether THIS surface's turn shows it.
784
+ fault: faultObsOf(scenario, toolCalls, status),
785
+ ...(typeof answer.content === 'string' ? { answer: answer.content } : {}),
786
+ ...(succeeded
787
+ ? {}
788
+ : { errorCode: answer.generationFailed ? 'generation_failed' : 'turn_failed' }),
789
+ noise: { at: Date.now() },
790
+ };
791
+ }
792
+ /**
793
+ * Collect one surface's tool-call lifecycle: the `called` phase only, which is
794
+ * the one that carries an outcome (`ok`). A call that never reached `called` did
795
+ * not happen, and an outcome that is absent is recorded as absent, not guessed.
796
+ */
797
+ function collectCalled(target, phase, info) {
798
+ if (phase !== 'called')
799
+ return;
800
+ target.push({ tool: info.tool, ...(typeof info.ok === 'boolean' ? { ok: info.ok } : {}) });
801
+ }
802
+ // ─── The drivers ────────────────────────────────────────────────────────────
803
+ /** CLI chat: the shared engine, called directly, pinned to the stub's model. */
804
+ async function runViaChatOnce(ws, scenario) {
805
+ const stub = await startStub(scenario);
806
+ try {
807
+ ws.useStub(stub.baseUrl);
808
+ await clearResponseCache();
809
+ const { ChatCommand } = await import('../cli/chat.js');
810
+ const command = new ChatCommand();
811
+ const toolCalls = [];
812
+ const { value: answer, otel } = await withOtlpCollector(() => command.answerOnce(scenario.message, {
813
+ provider: PARITY_PROVIDER_TYPE,
814
+ model: PARITY_MODEL,
815
+ onToolCall: (phase, info) => collectCalled(toolCalls, phase, info),
816
+ }));
817
+ return toObservation('cli-chat', scenario, { ...answer, debugLog: debugLogOf('cli-chat'), otel }, toolCalls, stub.chatCalls());
818
+ }
819
+ finally {
820
+ await stub.close();
821
+ }
822
+ }
823
+ /**
824
+ * Dashboard chat: the real console, with NO injected engine and no injected
825
+ * provider, so `ensureEngine()` lazily loads the real `ChatCommand` and the
826
+ * console's own turn plumbing (session record, busy guard, turn telemetry,
827
+ * progress emission) runs rather than being bypassed. The provider/model are
828
+ * pinned through the console's public options, which is exactly the surface
829
+ * handing the pin to its engine.
830
+ */
831
+ async function runViaConsole(ws, scenario) {
832
+ const stub = await startStub(scenario);
833
+ try {
834
+ ws.useStub(stub.baseUrl);
835
+ await clearResponseCache();
836
+ const { ChatConsole } = await import('../web-dashboard/chat-console.js');
837
+ const console_ = new ChatConsole({});
838
+ const toolCalls = [];
839
+ // The console's own subscription seam — the same one the GUI uses — rather
840
+ // than reaching into the engine, so this observes what a dashboard user sees.
841
+ const off = console_.onEvent((_sessionId, event) => {
842
+ if (event.kind !== 'tool')
843
+ return;
844
+ collectCalled(toolCalls, event.phase, {
845
+ tool: event.tool,
846
+ ...(event.ok === undefined ? {} : { ok: event.ok }),
847
+ });
848
+ });
849
+ try {
850
+ // WS5 (#27) — one session PER TURN, counter and all. The resume probe runs
851
+ // the same ask twice, and a second turn in the SAME conversation carries the
852
+ // first answer in its history — so its input genuinely differs, the replay
853
+ // correctly misses, and the row would compare two different questions. A
854
+ // fresh conversation each time is what the other surfaces do anyway (a
855
+ // one-shot CLI answer, a fresh child process). The counter keeps the ids
856
+ // unique within a run; nothing else reads them.
857
+ consoleRun += 1;
858
+ const { value: result, otel } = await withOtlpCollector(() => console_.answer(`parity-${scenario.id}-${consoleRun}`, scenario.message, {
859
+ provider: PARITY_PROVIDER_TYPE,
860
+ model: PARITY_MODEL,
861
+ }));
862
+ return toObservation('dashboard-chat', scenario, { ...result, debugLog: debugLogOf('dashboard-chat'), otel }, toolCalls, stub.chatCalls());
863
+ }
864
+ finally {
865
+ off();
866
+ }
867
+ }
868
+ finally {
869
+ await stub.close();
870
+ }
871
+ }
872
+ /** A channel adapter that records every send, so the gateway driver sees a real reply. */
873
+ class RecordingAdapter {
874
+ platform = 'mock';
875
+ configured = true;
876
+ sent = [];
877
+ describe() {
878
+ return 'Parity recorder';
879
+ }
880
+ async start() { }
881
+ async stop() { }
882
+ async send(_channelId, text) {
883
+ this.sent.push(text);
884
+ return true;
885
+ }
886
+ }
887
+ /** Unique per gateway run: the gateway dedups a re-delivered message. */
888
+ let gatewayRun = 0;
889
+ /**
890
+ * Unique per console turn (WS5): the resume probe drives the same ask twice, and
891
+ * two turns in one conversation would carry the first answer in the second's
892
+ * history (see the comment at the call site).
893
+ */
894
+ let consoleRun = 0;
895
+ /**
896
+ * Gateway (WhatsApp / Telegram / …): the REAL registry handler, with no chat
897
+ * engine injected, so `runInboundChat` lazily imports the real `ChatCommand`.
898
+ * The gateway derives the provider/model pair from its own config
899
+ * (`registry.ts:1797`) — which the harness has pointed at the stub — so this is
900
+ * the surface's own resolution, not something forced in from here.
901
+ *
902
+ * The observation is read from the gateway's OWN durable record, the
903
+ * `inbound.chat` log entry: a messaging surface has no terminal to scroll, so
904
+ * the log is where its attribution and tool lifecycle have to live. Reading the
905
+ * log rather than a return value is the honest test of that claim — if the
906
+ * gateway stops writing the triple, this goes red.
907
+ */
908
+ async function runViaGateway(ws, scenario) {
909
+ const stub = await startStub(scenario);
910
+ const deliveryDir = mkdtempSync(join(ws.root, 'gateway-'));
911
+ try {
912
+ ws.useStub(stub.baseUrl);
913
+ await clearResponseCache();
914
+ const { GatewayRegistry } = await import('../gateway/registry.js');
915
+ const adapter = new RecordingAdapter();
916
+ const registry = new GatewayRegistry({ streamEvents: false, deliveryConfigDir: deliveryDir });
917
+ registry.register(adapter);
918
+ gatewayRun += 1;
919
+ const channelId = `parity-${scenario.id}-${gatewayRun}`;
920
+ const { otel } = await withOtlpCollector(() => registry.handleInbound({
921
+ platform: 'mock',
922
+ channelId,
923
+ text: scenario.message,
924
+ from: 'parity',
925
+ senderId: 'parity',
926
+ }, { forceKind: 'chat' }));
927
+ const record = readGatewayLog(50).find((r) => r.event === 'inbound.chat' && r.channelId === channelId);
928
+ if (!record) {
929
+ throw new Error('the gateway did not record an inbound.chat turn for this message');
930
+ }
931
+ // The log carries the surface's own `{ tool, ok }` lifecycle; absence is
932
+ // preserved as absence (a call whose outcome frame never arrived).
933
+ const toolCalls = Array.isArray(record.toolCalls)
934
+ ? record.toolCalls.map((call) => ({
935
+ tool: call.tool,
936
+ ...(typeof call.ok === 'boolean' ? { ok: call.ok } : {}),
937
+ }))
938
+ : [];
939
+ const reply = [...adapter.sent].reverse().find((line) => line.includes(scenario.answer));
940
+ // WS1 — the findings the gateway recorded for this turn, read from its own
941
+ // durable record for the same reason the tool lifecycle is: a messaging
942
+ // surface has no terminal, so `inbound.chat` IS where this surface said it.
943
+ const findings = Array.isArray(record.findings)
944
+ ? record.findings
945
+ : [];
946
+ return toObservation('gateway-chat', scenario, {
947
+ content: reply,
948
+ findings,
949
+ provider: typeof record.provider === 'string' ? record.provider : undefined,
950
+ model: typeof record.model === 'string' ? record.model : undefined,
951
+ transport: record.transport === 'native' || record.transport === 'json'
952
+ ? record.transport
953
+ : record.transport === 'none'
954
+ ? 'none'
955
+ : undefined,
956
+ generationFailed: record.generationFailed === true,
957
+ // WS2 — read from the same isolated profile the gateway wrote into.
958
+ debugLog: debugLogOf('gateway-chat'),
959
+ // WS3 — the span tree the collector received from this surface.
960
+ otel,
961
+ // WS5 — read from the gateway's own durable record, for the same reason
962
+ // the tool lifecycle and the findings are: a messaging surface has no
963
+ // terminal, so `inbound.chat` IS where this surface said what it did.
964
+ ...(record.worktree && typeof record.worktree === 'object'
965
+ ? { worktree: record.worktree }
966
+ : {}),
967
+ ...(record.resume && typeof record.resume === 'object'
968
+ ? { resume: record.resume }
969
+ : {}),
970
+ }, toolCalls, stub.chatCalls());
971
+ }
972
+ finally {
973
+ await stub.close();
974
+ rmSync(deliveryDir, { recursive: true, force: true });
975
+ }
976
+ }
977
+ /**
978
+ * CLI execute / one-shot: the COMMAND's own single-goal path, with the provider
979
+ * served by the shared factory pointed at the stub. Both of the command's arms
980
+ * report now — the loop engine through `runLoopExecutor` and the direct chat
981
+ * answer through the engine's `onToolCall` — so the observation is what the
982
+ * COMMAND returns, not what the engine inside it happened to know.
983
+ *
984
+ * `runSingleGoal` is `private` in `src/cli/execute.ts`; the cast below is the
985
+ * same typed seam the tests already use, and it goes red the day the command
986
+ * stops handing its own result back.
987
+ */
988
+ async function runViaExecuteCommand(ws, scenario) {
989
+ const stub = await startStub(scenario);
990
+ try {
991
+ ws.useStub(stub.baseUrl);
992
+ await clearResponseCache();
993
+ const { ExecuteCommand } = await import('../cli/execute.js');
994
+ const command = new ExecuteCommand();
995
+ const { value: result, otel } = await withOtlpCollector(() => command.runSingleGoal(scenario.message, PARITY_PROVIDER_TYPE, PARITY_MODEL, {
996
+ engine: 'loop',
997
+ }));
998
+ return toObservation('cli-execute', scenario, {
999
+ content: result.content,
1000
+ provider: result.provider,
1001
+ model: result.model,
1002
+ transport: result.transport,
1003
+ generationFailed: !result.success,
1004
+ ...(result.findings ? { findings: result.findings } : {}),
1005
+ debugLog: debugLogOf('cli-execute'),
1006
+ // WS3 — the span tree this command's loop exported for the turn.
1007
+ otel,
1008
+ // WS5 — the command's own report, on either engine arm.
1009
+ ...(result.worktree ? { worktree: result.worktree } : {}),
1010
+ ...(result.resume ? { resume: result.resume } : {}),
1011
+ },
1012
+ // The command's own per-call outcomes (captured from the loop's
1013
+ // `tool`/`refusal` events). The fallback keeps a name-only result honest:
1014
+ // `ok` stays absent rather than being guessed.
1015
+ result.toolOutcomes
1016
+ ? result.toolOutcomes.map((call) => ({
1017
+ tool: call.tool,
1018
+ ...(typeof call.ok === 'boolean' ? { ok: call.ok } : {}),
1019
+ }))
1020
+ : (result.toolCalls ?? []).map((tool) => ({ tool })), stub.chatCalls());
1021
+ }
1022
+ finally {
1023
+ await stub.close();
1024
+ }
1025
+ }
1026
+ /**
1027
+ * Subagent: a REAL forked child process over its IPC channel, with the model
1028
+ * served by the same loopback stub. The child resolves its OWN provider from
1029
+ * config — that IS the isolation boundary — but the harness points that provider
1030
+ * at the same `groq` id and the same stub the in-process surfaces use, so
1031
+ * agreement is evidence rather than a stub coincidence.
1032
+ *
1033
+ * The observation is read from what the child itself reported (provider, model,
1034
+ * transport, llm calls, and each tool call + outcome on its progress frames),
1035
+ * never reconstructed here.
1036
+ */
1037
+ async function runViaSubagent(ws, scenario) {
1038
+ const stub = await startStub(scenario);
1039
+ // The child gets its OWN config/memory dirs, because it is its own process —
1040
+ // but the file it reads points at the SAME stub server.
1041
+ //
1042
+ // DETERMINISTIC, not `mkdtemp`: a resume probe drives the same scenario twice,
1043
+ // and the child's step record lives in the memory dir it inherits. A throwaway
1044
+ // dir per call would put the second turn's record in a different place from the
1045
+ // first one's, so the child could never replay anything and the probe would be
1046
+ // measuring the harness's own bookkeeping. Keyed by SCENARIO (not by surface) on
1047
+ // purpose: the child of one surface must not read another's record, and each
1048
+ // surface's own two turns must share one.
1049
+ const childRoot = join(ws.root, `subagent-${scenario.id}`);
1050
+ const childConfig = join(childRoot, 'config');
1051
+ mkdirSync(childConfig, { recursive: true });
1052
+ writeFileSync(join(childConfig, 'buffconfig.json'), JSON.stringify({
1053
+ defaultProvider: PARITY_PROVIDER_TYPE,
1054
+ providers: {
1055
+ [PARITY_PROVIDER_TYPE]: {
1056
+ apiKey: PARITY_API_KEY,
1057
+ model: PARITY_MODEL,
1058
+ baseUrl: stub.baseUrl,
1059
+ },
1060
+ },
1061
+ }));
1062
+ try {
1063
+ const { getSubagentManager } = await import('../tools/subagent-spawner.js');
1064
+ const manager = getSubagentManager();
1065
+ const callsByRun = new Map();
1066
+ // The listener is attached before the id exists, and a progress frame can
1067
+ // arrive before `spawn()` resolves — hence keying by run id.
1068
+ const onProgress = (id, msg) => {
1069
+ // The OUTCOME frame: the child reports `tool_call` then `tool_result`, and
1070
+ // the driver records the one that carries WHAT HAPPENED, exactly as the
1071
+ // `called` phase is read in-process. A missing outcome stays absent.
1072
+ if (msg?.phase !== 'tool_result' || typeof msg.tool !== 'string')
1073
+ return;
1074
+ const list = callsByRun.get(id) ?? [];
1075
+ list.push({ tool: msg.tool, ...(typeof msg.ok === 'boolean' ? { ok: msg.ok } : {}) });
1076
+ callsByRun.set(id, list);
1077
+ };
1078
+ manager.on('progress', onProgress);
1079
+ try {
1080
+ // A tool is ALWAYS offered to the child, even for a scenario that calls
1081
+ // none: the in-process surfaces always have their tools available, so
1082
+ // offering the child one keeps its transport attribution comparable. The
1083
+ // stub only asks for the tool when the scenario does.
1084
+ const availableTool = scenario.toolCall?.tool ?? 'list_dir';
1085
+ // WS3 — one collector for this surface, held across the WHOLE child run:
1086
+ // the fork inherits `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` (and `NUVIRA_OTEL`)
1087
+ // from this process's environment, and the child flushes into it before it
1088
+ // reports its result — so what is read here is the child's OWN export,
1089
+ // produced in its own process, rather than something reconstructed here.
1090
+ const spawnConfig = {
1091
+ goal: scenario.message,
1092
+ provider: PARITY_PROVIDER_TYPE,
1093
+ model: PARITY_MODEL,
1094
+ tools: [availableTool],
1095
+ // WS5 (#27) — delegation-level isolation: the PARENT makes the worktree,
1096
+ // forks the child INTO it and measures the diff itself. That is the
1097
+ // mechanism a delegating caller actually has (the child resolves its own
1098
+ // config and could decline), and it is why this surface is asked through
1099
+ // the spawn config rather than through the environment the in-process
1100
+ // surfaces read.
1101
+ ...(scenario.isolation === true ? { worktree: true } : {}),
1102
+ env: {
1103
+ NUVIRA_CONFIG_DIR: childConfig,
1104
+ NUVIRA_MEMORY_DIR: join(childRoot, 'memory'),
1105
+ // WS2 — the child keeps its OWN config/memory (that IS the isolation
1106
+ // boundary) but writes its debug log into the run's throwaway debug
1107
+ // dir, so this driver and a test can read the artifact the child
1108
+ // actually produced instead of re-deriving a path inside the child's
1109
+ // private profile.
1110
+ NUVIRA_DEBUG_LOG_DIR: debugLogDir(),
1111
+ },
1112
+ };
1113
+ const { value, otel } = await withOtlpCollector(async () => {
1114
+ const state = await manager.spawn(spawnConfig);
1115
+ try {
1116
+ return { state, result: await manager.waitForCompletion(state.id, 60_000) };
1117
+ }
1118
+ catch {
1119
+ // A CHILD THAT FAILED IS AN OBSERVATION, NOT A HARNESS CRASH — found by
1120
+ // the WS6 provider-fault row, which is the first scenario where the child
1121
+ // legitimately ends in failure. `waitForCompletion` REJECTS on a failed
1122
+ // run, so this throw used to escape the driver, abort `runParityScenario`
1123
+ // and fail the harness with an exception instead of a verdict. A harness
1124
+ // that cannot say "every surface failed honestly" cannot measure fault
1125
+ // handling at all.
1126
+ //
1127
+ // The rejection is DISCARDED in favour of the manager's own recorded
1128
+ // state: what is reported is the child's own report (its error, its
1129
+ // attribution, its call counts), not the exception this driver happened
1130
+ // to catch — the same "read the surface's own record" rule the gateway
1131
+ // and debug-log drivers follow.
1132
+ const failed = manager.getState(state.id) ?? state;
1133
+ return {
1134
+ state: failed,
1135
+ result: {
1136
+ id: failed.id,
1137
+ success: false,
1138
+ result: failed.result ?? '',
1139
+ ...(failed.error ? { error: failed.error } : {}),
1140
+ ...(failed.refusalCode ? { refusalCode: failed.refusalCode } : {}),
1141
+ ...(failed.provider ? { provider: failed.provider } : {}),
1142
+ ...(failed.model ? { model: failed.model } : {}),
1143
+ ...(failed.transport ? { transport: failed.transport } : {}),
1144
+ ...(failed.findings ? { findings: failed.findings } : {}),
1145
+ ...(failed.resume ? { resume: failed.resume } : {}),
1146
+ llmCalls: failed.llmCalls,
1147
+ tokensUsed: failed.tokensUsed,
1148
+ toolCalls: failed.toolCalls,
1149
+ durationMs: failed.durationMs ?? 0,
1150
+ log: manager.getLog(failed.id),
1151
+ },
1152
+ };
1153
+ }
1154
+ });
1155
+ const { state, result } = value;
1156
+ const transport = result.transport === 'native' || result.transport === 'json' ? result.transport : 'none';
1157
+ // The child's own verdict, through the shared helper (it reports success
1158
+ // directly, so `ok` is the flag it has) — the same rule across the fork
1159
+ // rather than a second one that could disagree.
1160
+ const status = turnStatus({ ok: result.success });
1161
+ const childCalls = callsByRun.get(state.id) ?? [];
1162
+ return {
1163
+ surface: 'subagent',
1164
+ engine: 'loop',
1165
+ status,
1166
+ modelCalls: result.llmCalls,
1167
+ ...(result.provider ? { provider: result.provider } : {}),
1168
+ ...(result.model ? { model: result.model } : {}),
1169
+ transport,
1170
+ toolCalls: childCalls,
1171
+ // WS1 — the child's own findings, read from the frames it sent.
1172
+ findings: result.findings ?? [],
1173
+ // WS2 — the child's own debug log, read back from the file it wrote
1174
+ // (its dir is pinned into the child's env above, so this is the CHILD's
1175
+ // artifact — produced in its own process, with its own provider object —
1176
+ // and not something reconstructed here).
1177
+ debugLog: debugLogOf('subagent'),
1178
+ // WS3 — the span tree the CHILD exported, read back from the collector it
1179
+ // was pointed at through its inherited environment.
1180
+ otel,
1181
+ // WS4 — replaced by the driver wrapper with what the operator's declared
1182
+ // hook received: the child runs the hook in its OWN process and appends
1183
+ // to the log file it inherited the path of (`withToolHooks`).
1184
+ hooks: noToolHooks(),
1185
+ // WS5 — the isolation the PARENT made for this child, and what the diff
1186
+ // against the base commit was. Read from the result the manager hands its
1187
+ // caller, which is the artifact a delegating caller actually receives.
1188
+ isolation: isolationObsOf(scenario.isolation === true, result.worktree),
1189
+ // WS5 — the CHILD's own resume report, read from the frame it sent (the
1190
+ // child is a separate process; this is its only channel back).
1191
+ resume: resumeObsOf(scenario.resume === true, result.resume),
1192
+ // WS6 (#28) — the declared fault, derived from what the CHILD's own frames
1193
+ // and result say. This is the field that proves a declaration crosses the
1194
+ // fork: a `tool` fault reaches this child through the environment the
1195
+ // parent handed it, and the failed call is reported back on a frame.
1196
+ fault: faultObsOf(scenario, childCalls, status),
1197
+ ...(result.result ? { answer: result.result } : {}),
1198
+ ...(result.success ? {} : { errorCode: result.refusalCode ?? 'turn_failed' }),
1199
+ noise: { at: Date.now() },
1200
+ };
1201
+ }
1202
+ finally {
1203
+ manager.off('progress', onProgress);
1204
+ }
1205
+ }
1206
+ finally {
1207
+ // The child's root is left in place (the workspace's own teardown removes it):
1208
+ // a resume probe's two turns must share it, so deleting it here would delete the
1209
+ // record the second turn is about to read.
1210
+ await stub.close();
1211
+ }
1212
+ }
1213
+ /**
1214
+ * A driver that exists only to record why it cannot run. Exported so a caller
1215
+ * can pin the guard that a blocked surface is never silently invoked — the real
1216
+ * driver list currently needs no blocks.
1217
+ */
1218
+ export function blockedDriver(surface, reason) {
1219
+ return {
1220
+ surface,
1221
+ depth: DRIVER_DEPTH,
1222
+ available: false,
1223
+ blockedBy: reason,
1224
+ run: async () => {
1225
+ throw new Error(`parity driver for ${surface} is blocked and must never be invoked: ${reason}`);
1226
+ },
1227
+ };
1228
+ }
1229
+ /**
1230
+ * Build the real drivers.
1231
+ *
1232
+ * The environment is switched to a throwaway profile here and restored in
1233
+ * `dispose()`. Every module that resolves `NUVIRA_CONFIG_DIR` / `NUVIRA_MEMORY_DIR`
1234
+ * at call time (the config manager, the cache, the gateway log) therefore reads
1235
+ * the isolated profile for the whole run, and the developer's real profile is
1236
+ * untouched — the same hermetic convention the test suite uses.
1237
+ */
1238
+ export async function createParityHarness() {
1239
+ const workspace = createWorkspace();
1240
+ const previous = {
1241
+ configDir: process.env.NUVIRA_CONFIG_DIR,
1242
+ memoryDir: process.env.NUVIRA_MEMORY_DIR,
1243
+ buffConfigDir: process.env.BUFF_CONFIG_DIR,
1244
+ buffMemoryDir: process.env.BUFF_MEMORY_DIR,
1245
+ debugLog: process.env.NUVIRA_DEBUG_LOG,
1246
+ otel: process.env[otelEnableVarName],
1247
+ otelEndpoint: process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT,
1248
+ // WS5 (#27) — never the developer's own request. The harness DECLARES these
1249
+ // per scenario (`withTurnEnvelope`); a value inherited from the shell would
1250
+ // make the scenarios that do not ask for the capability measure it anyway.
1251
+ isolation: process.env[WORKTREE_ENABLE_ENV],
1252
+ resume: process.env[RESUME_ENABLE_ENV],
1253
+ // WS6 (#28) — same rule: the harness DECLARES a fault per scenario, so a
1254
+ // declaration inherited from the shell must not arm one for every scenario.
1255
+ fault: process.env[FAULT_ENV],
1256
+ };
1257
+ process.env.NUVIRA_CONFIG_DIR = workspace.configDir;
1258
+ process.env.NUVIRA_MEMORY_DIR = workspace.memoryDir;
1259
+ // WS2 — logging is turned ON for the whole run, so "this surface wrote a log
1260
+ // whose header names the backend" is an ASSERTION rather than something the
1261
+ // harness never asked for. Every surface writes into the isolated profile
1262
+ // above (a forked child into its own), so nothing reaches the developer's
1263
+ // real `~/.nuvira`.
1264
+ process.env.NUVIRA_DEBUG_LOG = '1';
1265
+ // WS3 — span export is turned ON for the whole run, so "this surface exported
1266
+ // its turn" is an ASSERTION rather than something the harness never asked for.
1267
+ // Only the GATE belongs here; the endpoint is set per driver
1268
+ // (`withOtlpCollector`), because each surface gets its own collector.
1269
+ process.env[otelEnableVarName] = '1';
1270
+ // WS6 (#28) — and no fault, until a scenario declares one.
1271
+ delete process.env[FAULT_ENV];
1272
+ resetFaultInjector();
1273
+ // The legacy aliases would otherwise win on the modules that check them, and
1274
+ // point half the run back at the developer's real profile.
1275
+ delete process.env.BUFF_CONFIG_DIR;
1276
+ delete process.env.BUFF_MEMORY_DIR;
1277
+ const driver = (surface, run) => ({
1278
+ surface,
1279
+ depth: DRIVER_DEPTH,
1280
+ available: true,
1281
+ run: async (scenario) => {
1282
+ // WS3 — the SDK keeps ONE provider per process and reads the endpoint when
1283
+ // it builds it, so the previous surface's provider (pointing at the
1284
+ // previous collector, now closed) has to go before this surface starts.
1285
+ // Without this reset the second surface would export into the first
1286
+ // surface's collector, and every surface after that would read as having
1287
+ // exported nothing — a silent, permanent green on one surface only.
1288
+ await shutdownSpans();
1289
+ // WS5 — isolation and resume are DECLARED around the turn (and undeclared
1290
+ // again after it), so "this surface isolated its turn / replayed its steps"
1291
+ // is an assertion about a declaration the harness actually made. OUTSIDE
1292
+ // the hook wrapper because the resume probe runs the turn twice, and each
1293
+ // of those two turns is a whole turn of its own.
1294
+ return withTurnEnvelope(scenario, surface, () =>
1295
+ // WS4 — the operator's hooks are DECLARED around the turn (and undeclared
1296
+ // again after it), so "this surface ran the hook" is an assertion about a
1297
+ // declaration the harness actually made.
1298
+ withToolHooks(workspace, surface, scenario, () => run(workspace, scenario)));
1299
+ },
1300
+ });
1301
+ const inProcess = [
1302
+ driver('cli-chat', runViaChatOnce),
1303
+ driver('dashboard-chat', runViaConsole),
1304
+ driver('gateway-chat', runViaGateway),
1305
+ driver('cli-execute', runViaExecuteCommand),
1306
+ ];
1307
+ const subagent = [driver('subagent', runViaSubagent)];
1308
+ return {
1309
+ drivers: [...inProcess, ...subagent],
1310
+ inProcess,
1311
+ subagent,
1312
+ async dispose() {
1313
+ if (previous.configDir === undefined)
1314
+ delete process.env.NUVIRA_CONFIG_DIR;
1315
+ else
1316
+ process.env.NUVIRA_CONFIG_DIR = previous.configDir;
1317
+ if (previous.memoryDir === undefined)
1318
+ delete process.env.NUVIRA_MEMORY_DIR;
1319
+ else
1320
+ process.env.NUVIRA_MEMORY_DIR = previous.memoryDir;
1321
+ if (previous.buffConfigDir === undefined)
1322
+ delete process.env.BUFF_CONFIG_DIR;
1323
+ else
1324
+ process.env.BUFF_CONFIG_DIR = previous.buffConfigDir;
1325
+ if (previous.buffMemoryDir === undefined)
1326
+ delete process.env.BUFF_MEMORY_DIR;
1327
+ else
1328
+ process.env.BUFF_MEMORY_DIR = previous.buffMemoryDir;
1329
+ if (previous.debugLog === undefined)
1330
+ delete process.env.NUVIRA_DEBUG_LOG;
1331
+ else
1332
+ process.env.NUVIRA_DEBUG_LOG = previous.debugLog;
1333
+ if (previous.otel === undefined)
1334
+ delete process.env[otelEnableVarName];
1335
+ else
1336
+ process.env[otelEnableVarName] = previous.otel;
1337
+ if (previous.otelEndpoint === undefined)
1338
+ delete process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT;
1339
+ else
1340
+ process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = previous.otelEndpoint;
1341
+ if (previous.isolation === undefined)
1342
+ delete process.env[WORKTREE_ENABLE_ENV];
1343
+ else
1344
+ process.env[WORKTREE_ENABLE_ENV] = previous.isolation;
1345
+ if (previous.resume === undefined)
1346
+ delete process.env[RESUME_ENABLE_ENV];
1347
+ else
1348
+ process.env[RESUME_ENABLE_ENV] = previous.resume;
1349
+ if (previous.fault === undefined)
1350
+ delete process.env[FAULT_ENV];
1351
+ else
1352
+ process.env[FAULT_ENV] = previous.fault;
1353
+ resetFaultInjector();
1354
+ // Tear the LAST surface's provider down too: a test file that runs several
1355
+ // parity runs in a row would otherwise inherit the first one's provider,
1356
+ // still pointing at a collector that has been closed.
1357
+ await shutdownSpans();
1358
+ rmSync(workspace.root, { recursive: true, force: true });
1359
+ },
1360
+ };
1361
+ }
1362
+ //# sourceMappingURL=drivers.js.map