agent-nuvira 3.3.3 → 3.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (349) hide show
  1. package/README.md +13 -5
  2. package/dist/agent-sdk/src/agent.d.ts +2 -0
  3. package/dist/agent-sdk/src/agent.d.ts.map +1 -1
  4. package/dist/agent-sdk/src/define.d.ts +64 -0
  5. package/dist/agent-sdk/src/define.d.ts.map +1 -0
  6. package/dist/agent-sdk/src/define.js +76 -0
  7. package/dist/agent-sdk/src/define.js.map +1 -0
  8. package/dist/agent-sdk/src/index.d.ts +9 -0
  9. package/dist/agent-sdk/src/index.d.ts.map +1 -1
  10. package/dist/agent-sdk/src/index.js +9 -0
  11. package/dist/agent-sdk/src/index.js.map +1 -1
  12. package/dist/agent-sdk/src/scaffold.d.ts +11 -0
  13. package/dist/agent-sdk/src/scaffold.d.ts.map +1 -1
  14. package/dist/agent-sdk/src/scaffold.js +16 -6
  15. package/dist/agent-sdk/src/scaffold.js.map +1 -1
  16. package/dist/agents/long-form-plan.d.ts.map +1 -1
  17. package/dist/agents/long-form-plan.js +2 -1
  18. package/dist/agents/long-form-plan.js.map +1 -1
  19. package/dist/agents/orchestrator.d.ts.map +1 -1
  20. package/dist/agents/orchestrator.js +5 -4
  21. package/dist/agents/orchestrator.js.map +1 -1
  22. package/dist/cli/agent.d.ts +2 -2
  23. package/dist/cli/agent.js +10 -10
  24. package/dist/cli/chat.d.ts +94 -0
  25. package/dist/cli/chat.d.ts.map +1 -1
  26. package/dist/cli/chat.js +354 -20
  27. package/dist/cli/chat.js.map +1 -1
  28. package/dist/cli/cli-program.d.ts.map +1 -1
  29. package/dist/cli/cli-program.js +5 -0
  30. package/dist/cli/cli-program.js.map +1 -1
  31. package/dist/cli/config.d.ts.map +1 -1
  32. package/dist/cli/config.js +9 -1
  33. package/dist/cli/config.js.map +1 -1
  34. package/dist/cli/doctor.d.ts.map +1 -1
  35. package/dist/cli/doctor.js +3 -2
  36. package/dist/cli/doctor.js.map +1 -1
  37. package/dist/cli/edit.js +2 -2
  38. package/dist/cli/eval.d.ts +17 -0
  39. package/dist/cli/eval.d.ts.map +1 -1
  40. package/dist/cli/eval.js +105 -2
  41. package/dist/cli/eval.js.map +1 -1
  42. package/dist/cli/execute.d.ts +16 -1
  43. package/dist/cli/execute.d.ts.map +1 -1
  44. package/dist/cli/execute.js +192 -22
  45. package/dist/cli/execute.js.map +1 -1
  46. package/dist/cli/loop-executor.d.ts +55 -0
  47. package/dist/cli/loop-executor.d.ts.map +1 -1
  48. package/dist/cli/loop-executor.js +212 -13
  49. package/dist/cli/loop-executor.js.map +1 -1
  50. package/dist/cli/model.d.ts.map +1 -1
  51. package/dist/cli/model.js +9 -8
  52. package/dist/cli/model.js.map +1 -1
  53. package/dist/cli/models.d.ts +1 -1
  54. package/dist/cli/models.js +4 -4
  55. package/dist/cli/parity.d.ts +85 -0
  56. package/dist/cli/parity.d.ts.map +1 -0
  57. package/dist/cli/parity.js +506 -0
  58. package/dist/cli/parity.js.map +1 -0
  59. package/dist/cli/plan.d.ts.map +1 -1
  60. package/dist/cli/plan.js +2 -1
  61. package/dist/cli/plan.js.map +1 -1
  62. package/dist/cli/retrieval.d.ts.map +1 -1
  63. package/dist/cli/retrieval.js +5 -4
  64. package/dist/cli/retrieval.js.map +1 -1
  65. package/dist/cli/sdk.js +4 -4
  66. package/dist/cli/sdk.js.map +1 -1
  67. package/dist/cli/trace.d.ts.map +1 -1
  68. package/dist/cli/trace.js +2 -1
  69. package/dist/cli/trace.js.map +1 -1
  70. package/dist/cli/workflow.js +2 -2
  71. package/dist/cli/workflow.js.map +1 -1
  72. package/dist/config/types.d.ts +44 -0
  73. package/dist/config/types.d.ts.map +1 -1
  74. package/dist/findings/verdicts.d.ts +241 -0
  75. package/dist/findings/verdicts.d.ts.map +1 -0
  76. package/dist/findings/verdicts.js +284 -0
  77. package/dist/findings/verdicts.js.map +1 -0
  78. package/dist/gateway/adapters.d.ts +65 -0
  79. package/dist/gateway/adapters.d.ts.map +1 -1
  80. package/dist/gateway/adapters.js +216 -10
  81. package/dist/gateway/adapters.js.map +1 -1
  82. package/dist/gateway/channel-directory.d.ts +31 -0
  83. package/dist/gateway/channel-directory.d.ts.map +1 -1
  84. package/dist/gateway/channel-directory.js +40 -0
  85. package/dist/gateway/channel-directory.js.map +1 -1
  86. package/dist/gateway/gateway-log.d.ts +1 -1
  87. package/dist/gateway/gateway-log.d.ts.map +1 -1
  88. package/dist/gateway/gateway-log.js.map +1 -1
  89. package/dist/gateway/hooks.d.ts +87 -19
  90. package/dist/gateway/hooks.d.ts.map +1 -1
  91. package/dist/gateway/hooks.js +62 -23
  92. package/dist/gateway/hooks.js.map +1 -1
  93. package/dist/gateway/inbound-media.d.ts +147 -0
  94. package/dist/gateway/inbound-media.d.ts.map +1 -0
  95. package/dist/gateway/inbound-media.js +317 -0
  96. package/dist/gateway/inbound-media.js.map +1 -0
  97. package/dist/gateway/inbox.d.ts +8 -1
  98. package/dist/gateway/inbox.d.ts.map +1 -1
  99. package/dist/gateway/inbox.js.map +1 -1
  100. package/dist/gateway/platform-config.d.ts +14 -0
  101. package/dist/gateway/platform-config.d.ts.map +1 -1
  102. package/dist/gateway/platform-config.js +26 -8
  103. package/dist/gateway/platform-config.js.map +1 -1
  104. package/dist/gateway/realtime.d.ts +114 -0
  105. package/dist/gateway/realtime.d.ts.map +1 -0
  106. package/dist/gateway/realtime.js +402 -0
  107. package/dist/gateway/realtime.js.map +1 -0
  108. package/dist/gateway/registry.d.ts +31 -0
  109. package/dist/gateway/registry.d.ts.map +1 -1
  110. package/dist/gateway/registry.js +224 -4
  111. package/dist/gateway/registry.js.map +1 -1
  112. package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
  113. package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
  114. package/dist/gateway/whatsapp/baileys-bridge.js +123 -4
  115. package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
  116. package/dist/gateway/whatsapp/bridge.d.ts +6 -2
  117. package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
  118. package/dist/gateway/whatsapp/bridge.js.map +1 -1
  119. package/dist/index.js +8 -0
  120. package/dist/index.js.map +1 -1
  121. package/dist/inference/factory.d.ts +14 -0
  122. package/dist/inference/factory.d.ts.map +1 -1
  123. package/dist/inference/factory.js +17 -0
  124. package/dist/inference/factory.js.map +1 -1
  125. package/dist/inference/groq-adapter.d.ts +2 -0
  126. package/dist/inference/groq-adapter.d.ts.map +1 -1
  127. package/dist/inference/groq-adapter.js +16 -6
  128. package/dist/inference/groq-adapter.js.map +1 -1
  129. package/dist/inference/tools.d.ts.map +1 -1
  130. package/dist/inference/tools.js +29 -0
  131. package/dist/inference/tools.js.map +1 -1
  132. package/dist/learning/benchmark.d.ts.map +1 -1
  133. package/dist/learning/benchmark.js +3 -2
  134. package/dist/learning/benchmark.js.map +1 -1
  135. package/dist/learning/continuation.d.ts.map +1 -1
  136. package/dist/learning/continuation.js +2 -1
  137. package/dist/learning/continuation.js.map +1 -1
  138. package/dist/learning/cost-tracker.d.ts.map +1 -1
  139. package/dist/learning/cost-tracker.js +2 -1
  140. package/dist/learning/cost-tracker.js.map +1 -1
  141. package/dist/learning/deferred-task.d.ts.map +1 -1
  142. package/dist/learning/deferred-task.js +13 -4
  143. package/dist/learning/deferred-task.js.map +1 -1
  144. package/dist/learning/eval-framework.d.ts.map +1 -1
  145. package/dist/learning/eval-framework.js +2 -1
  146. package/dist/learning/eval-framework.js.map +1 -1
  147. package/dist/learning/long-form.d.ts.map +1 -1
  148. package/dist/learning/long-form.js +2 -1
  149. package/dist/learning/long-form.js.map +1 -1
  150. package/dist/learning/model-registry.d.ts.map +1 -1
  151. package/dist/learning/model-registry.js +2 -1
  152. package/dist/learning/model-registry.js.map +1 -1
  153. package/dist/learning/reasoning-cache.d.ts.map +1 -1
  154. package/dist/learning/reasoning-cache.js +2 -1
  155. package/dist/learning/reasoning-cache.js.map +1 -1
  156. package/dist/learning/reasoning-trace.d.ts +37 -1
  157. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  158. package/dist/learning/reasoning-trace.js +66 -0
  159. package/dist/learning/reasoning-trace.js.map +1 -1
  160. package/dist/learning/resilient-call.d.ts.map +1 -1
  161. package/dist/learning/resilient-call.js +2 -1
  162. package/dist/learning/resilient-call.js.map +1 -1
  163. package/dist/learning/retrieval.d.ts.map +1 -1
  164. package/dist/learning/retrieval.js +2 -1
  165. package/dist/learning/retrieval.js.map +1 -1
  166. package/dist/learning/seeded-benchmark.d.ts +160 -0
  167. package/dist/learning/seeded-benchmark.d.ts.map +1 -0
  168. package/dist/learning/seeded-benchmark.js +321 -0
  169. package/dist/learning/seeded-benchmark.js.map +1 -0
  170. package/dist/learning/seeded-bugs.d.ts +142 -0
  171. package/dist/learning/seeded-bugs.d.ts.map +1 -0
  172. package/dist/learning/seeded-bugs.js +535 -0
  173. package/dist/learning/seeded-bugs.js.map +1 -0
  174. package/dist/learning/step-checkpoint.d.ts +127 -0
  175. package/dist/learning/step-checkpoint.d.ts.map +1 -0
  176. package/dist/learning/step-checkpoint.js +244 -0
  177. package/dist/learning/step-checkpoint.js.map +1 -0
  178. package/dist/nlu/intent-confirm.d.ts +23 -1
  179. package/dist/nlu/intent-confirm.d.ts.map +1 -1
  180. package/dist/nlu/intent-confirm.js +85 -2
  181. package/dist/nlu/intent-confirm.js.map +1 -1
  182. package/dist/nlu/learnings.d.ts +7 -0
  183. package/dist/nlu/learnings.d.ts.map +1 -1
  184. package/dist/nlu/learnings.js +7 -0
  185. package/dist/nlu/learnings.js.map +1 -1
  186. package/dist/observability/debug-log.d.ts +250 -0
  187. package/dist/observability/debug-log.d.ts.map +1 -0
  188. package/dist/observability/debug-log.js +500 -0
  189. package/dist/observability/debug-log.js.map +1 -0
  190. package/dist/observability/event-bus.d.ts.map +1 -1
  191. package/dist/observability/event-bus.js +4 -1
  192. package/dist/observability/event-bus.js.map +1 -1
  193. package/dist/observability/otel.d.ts +278 -0
  194. package/dist/observability/otel.d.ts.map +1 -0
  195. package/dist/observability/otel.js +590 -0
  196. package/dist/observability/otel.js.map +1 -0
  197. package/dist/parity/drivers.d.ts +99 -0
  198. package/dist/parity/drivers.d.ts.map +1 -0
  199. package/dist/parity/drivers.js +1362 -0
  200. package/dist/parity/drivers.js.map +1 -0
  201. package/dist/parity/graph.d.ts +73 -0
  202. package/dist/parity/graph.d.ts.map +1 -0
  203. package/dist/parity/graph.js +162 -0
  204. package/dist/parity/graph.js.map +1 -0
  205. package/dist/parity/matrix.d.ts +105 -0
  206. package/dist/parity/matrix.d.ts.map +1 -0
  207. package/dist/parity/matrix.js +352 -0
  208. package/dist/parity/matrix.js.map +1 -0
  209. package/dist/parity/observation.d.ts +444 -0
  210. package/dist/parity/observation.d.ts.map +1 -0
  211. package/dist/parity/observation.js +333 -0
  212. package/dist/parity/observation.js.map +1 -0
  213. package/dist/parity/scenarios.d.ts +229 -0
  214. package/dist/parity/scenarios.d.ts.map +1 -0
  215. package/dist/parity/scenarios.js +175 -0
  216. package/dist/parity/scenarios.js.map +1 -0
  217. package/dist/parity/surfaces.d.ts +122 -0
  218. package/dist/parity/surfaces.d.ts.map +1 -0
  219. package/dist/parity/surfaces.js +190 -0
  220. package/dist/parity/surfaces.js.map +1 -0
  221. package/dist/runtime/fault-injection.d.ts +173 -0
  222. package/dist/runtime/fault-injection.d.ts.map +1 -0
  223. package/dist/runtime/fault-injection.js +281 -0
  224. package/dist/runtime/fault-injection.js.map +1 -0
  225. package/dist/tools/child-agent-entry.d.ts +23 -0
  226. package/dist/tools/child-agent-entry.d.ts.map +1 -0
  227. package/dist/tools/child-agent-entry.js +129 -0
  228. package/dist/tools/child-agent-entry.js.map +1 -0
  229. package/dist/tools/child-agent-runtime.d.ts +124 -0
  230. package/dist/tools/child-agent-runtime.d.ts.map +1 -0
  231. package/dist/tools/child-agent-runtime.js +704 -0
  232. package/dist/tools/child-agent-runtime.js.map +1 -0
  233. package/dist/tools/coding-tools.d.ts.map +1 -1
  234. package/dist/tools/coding-tools.js +82 -10
  235. package/dist/tools/coding-tools.js.map +1 -1
  236. package/dist/tools/delegation-system.d.ts +31 -0
  237. package/dist/tools/delegation-system.d.ts.map +1 -1
  238. package/dist/tools/delegation-system.js +70 -9
  239. package/dist/tools/delegation-system.js.map +1 -1
  240. package/dist/tools/extract/docx.d.ts +27 -0
  241. package/dist/tools/extract/docx.d.ts.map +1 -0
  242. package/dist/tools/extract/docx.js +48 -0
  243. package/dist/tools/extract/docx.js.map +1 -0
  244. package/dist/tools/extract/html-text.d.ts +27 -0
  245. package/dist/tools/extract/html-text.d.ts.map +1 -0
  246. package/dist/tools/extract/html-text.js +86 -0
  247. package/dist/tools/extract/html-text.js.map +1 -0
  248. package/dist/tools/extract/pdf-ocr.d.ts +58 -0
  249. package/dist/tools/extract/pdf-ocr.d.ts.map +1 -0
  250. package/dist/tools/extract/pdf-ocr.js +116 -0
  251. package/dist/tools/extract/pdf-ocr.js.map +1 -0
  252. package/dist/tools/extract/pdf.d.ts +65 -0
  253. package/dist/tools/extract/pdf.d.ts.map +1 -0
  254. package/dist/tools/extract/pdf.js +197 -0
  255. package/dist/tools/extract/pdf.js.map +1 -0
  256. package/dist/tools/extract/pptx.d.ts +32 -0
  257. package/dist/tools/extract/pptx.d.ts.map +1 -0
  258. package/dist/tools/extract/pptx.js +77 -0
  259. package/dist/tools/extract/pptx.js.map +1 -0
  260. package/dist/tools/extract/xlsx.d.ts +47 -0
  261. package/dist/tools/extract/xlsx.d.ts.map +1 -0
  262. package/dist/tools/extract/xlsx.js +111 -0
  263. package/dist/tools/extract/xlsx.js.map +1 -0
  264. package/dist/tools/finding-tool.d.ts +76 -0
  265. package/dist/tools/finding-tool.d.ts.map +1 -0
  266. package/dist/tools/finding-tool.js +125 -0
  267. package/dist/tools/finding-tool.js.map +1 -0
  268. package/dist/tools/messaging-tools.d.ts +41 -11
  269. package/dist/tools/messaging-tools.d.ts.map +1 -1
  270. package/dist/tools/messaging-tools.js +104 -55
  271. package/dist/tools/messaging-tools.js.map +1 -1
  272. package/dist/tools/neutts-synth.d.ts +27 -3
  273. package/dist/tools/neutts-synth.d.ts.map +1 -1
  274. package/dist/tools/neutts-synth.js +57 -13
  275. package/dist/tools/neutts-synth.js.map +1 -1
  276. package/dist/tools/pipeline-tool.d.ts +43 -1
  277. package/dist/tools/pipeline-tool.d.ts.map +1 -1
  278. package/dist/tools/pipeline-tool.js +13 -2
  279. package/dist/tools/pipeline-tool.js.map +1 -1
  280. package/dist/tools/read-extract.d.ts +116 -45
  281. package/dist/tools/read-extract.d.ts.map +1 -1
  282. package/dist/tools/read-extract.js +494 -158
  283. package/dist/tools/read-extract.js.map +1 -1
  284. package/dist/tools/registry.d.ts +2 -2
  285. package/dist/tools/registry.d.ts.map +1 -1
  286. package/dist/tools/registry.js +152 -23
  287. package/dist/tools/registry.js.map +1 -1
  288. package/dist/tools/subagent-refusal.d.ts +15 -0
  289. package/dist/tools/subagent-refusal.d.ts.map +1 -0
  290. package/dist/tools/subagent-refusal.js +18 -0
  291. package/dist/tools/subagent-refusal.js.map +1 -0
  292. package/dist/tools/subagent-spawner.d.ts +122 -0
  293. package/dist/tools/subagent-spawner.d.ts.map +1 -1
  294. package/dist/tools/subagent-spawner.js +249 -28
  295. package/dist/tools/subagent-spawner.js.map +1 -1
  296. package/dist/tools/tool-hooks.d.ts +177 -0
  297. package/dist/tools/tool-hooks.d.ts.map +1 -0
  298. package/dist/tools/tool-hooks.js +427 -0
  299. package/dist/tools/tool-hooks.js.map +1 -0
  300. package/dist/tools/tool-loop.d.ts +82 -0
  301. package/dist/tools/tool-loop.d.ts.map +1 -1
  302. package/dist/tools/tool-loop.js +167 -9
  303. package/dist/tools/tool-loop.js.map +1 -1
  304. package/dist/tools/tool-refusal.d.ts +68 -0
  305. package/dist/tools/tool-refusal.d.ts.map +1 -0
  306. package/dist/tools/tool-refusal.js +78 -0
  307. package/dist/tools/tool-refusal.js.map +1 -0
  308. package/dist/tools/toolsets.d.ts +8 -0
  309. package/dist/tools/toolsets.d.ts.map +1 -1
  310. package/dist/tools/toolsets.js +12 -2
  311. package/dist/tools/toolsets.js.map +1 -1
  312. package/dist/tools/vision-tools.d.ts +88 -83
  313. package/dist/tools/vision-tools.d.ts.map +1 -1
  314. package/dist/tools/vision-tools.js +134 -103
  315. package/dist/tools/vision-tools.js.map +1 -1
  316. package/dist/tools/worktree.d.ts +210 -0
  317. package/dist/tools/worktree.d.ts.map +1 -0
  318. package/dist/tools/worktree.js +374 -0
  319. package/dist/tools/worktree.js.map +1 -0
  320. package/dist/utils/format.d.ts +3 -0
  321. package/dist/utils/format.d.ts.map +1 -0
  322. package/dist/utils/format.js +32 -0
  323. package/dist/utils/format.js.map +1 -0
  324. package/dist/web-dashboard/attachment-extract.d.ts +64 -0
  325. package/dist/web-dashboard/attachment-extract.d.ts.map +1 -0
  326. package/dist/web-dashboard/attachment-extract.js +154 -0
  327. package/dist/web-dashboard/attachment-extract.js.map +1 -0
  328. package/dist/web-dashboard/chat-console.d.ts +108 -1
  329. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  330. package/dist/web-dashboard/chat-console.js +36 -0
  331. package/dist/web-dashboard/chat-console.js.map +1 -1
  332. package/dist/web-dashboard/hub-data.d.ts +37 -0
  333. package/dist/web-dashboard/hub-data.d.ts.map +1 -1
  334. package/dist/web-dashboard/hub-data.js +61 -1
  335. package/dist/web-dashboard/hub-data.js.map +1 -1
  336. package/dist/web-dashboard/server.d.ts +14 -0
  337. package/dist/web-dashboard/server.d.ts.map +1 -1
  338. package/dist/web-dashboard/server.js +247 -13
  339. package/dist/web-dashboard/server.js.map +1 -1
  340. package/dist/web-dashboard/src/types.d.ts +196 -48
  341. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  342. package/package.json +18 -3
  343. package/src/web-dashboard/public/assets/{index-Cyd6tIew.css → index-XJjj2cBX.css} +1 -1
  344. package/src/web-dashboard/public/assets/index-YWF9FpwQ.js +207 -0
  345. package/src/web-dashboard/public/assets/index-YWF9FpwQ.js.map +1 -0
  346. package/src/web-dashboard/public/index.html +2 -2
  347. package/dist/tools/child-agent-worker.js +0 -212
  348. package/src/web-dashboard/public/assets/index-CxDj7p6i.js +0 -207
  349. package/src/web-dashboard/public/assets/index-CxDj7p6i.js.map +0 -1
package/dist/cli/chat.js CHANGED
@@ -15,6 +15,12 @@ import { maybeAutoRecall, recallContextBlock } from '../context/session-recall.j
15
15
  import { getMemoryManager } from '../memory/manager.js';
16
16
  import { logger } from '../utils/logger.js';
17
17
  import { printOrchestrationResult } from './execute.js';
18
+ // WS5 (#27) — isolation (a git worktree around the turn) and resume (replaying
19
+ // recorded steps instead of re-paying for them). See the module headers for why
20
+ // the worktree is created HERE, around the whole turn, and why a replay is
21
+ // keyed on the step's whole input rather than its position alone.
22
+ import { beginIsolation, endIsolation, resolveIsolationRequest, worktreeNotice, } from '../tools/worktree.js';
23
+ import { closeResume, openResume, resolveResumeRequest, } from '../learning/step-checkpoint.js';
18
24
  import { applyActiveModel } from './model.js';
19
25
  import { getProviderFallback, classifyFallbackError, isRetryableError, isTransientForRetry, recordRegistrySuccess } from '../learning/provider-fallback.js';
20
26
  import { recordActionFailure } from '../learning/failure-bookkeeping.js';
@@ -35,8 +41,10 @@ import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
35
41
  import { resolveDispatch } from '../nlu/actions.js';
36
42
  import { hasCodingAction, resolveAskKind } from '../nlu/conversation-gate.js';
37
43
  import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
44
+ // WS3 (#25) — the turn as a span, when an operator has asked for OTLP export.
45
+ import { flushSpans, otelNoticeOnce, startTurnSpan } from '../observability/otel.js';
38
46
  import { detectAnswerQualityFailure, answerQualityError, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
39
- import { beginTrace, endTrace, recordStep, recordTraceEvent, buildTraceOutcome } from '../learning/reasoning-trace.js';
47
+ import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, buildTraceOutcome } from '../learning/reasoning-trace.js';
40
48
  import { recordWorkingState, getWorkingState, formatWorkingState } from '../learning/working-state.js';
41
49
  import { getLoopExposureMode } from '../tools/toolsets.js';
42
50
  import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
@@ -46,6 +54,9 @@ import { sweepTransientFailures, collectionRevivalStore } from '../learning/prov
46
54
  import { analyzeComplexity } from '../learning/hybrid-router.js';
47
55
  import { routingCacheSignature, withRoutingCache } from '../learning/routing-cache.js';
48
56
  import { getTool, TOOL_CONTRACT_JSON } from '../tools/registry.js';
57
+ // WS1 — the finding tool's bus event, and the wire shape every surface reports.
58
+ import { FINDING_EVENT } from '../tools/finding-tool.js';
59
+ import { debugLogNotice, sessionDebugLog } from '../observability/debug-log.js';
49
60
  import { buildFollowupContinuationPrompt, isSuggestedFollowup, } from '../tools/followup-utils.js';
50
61
  // S2/S3 — the shared tool-call reliability helpers (salvage failed_generation,
51
62
  // compact fallback schemas). One copy for every tool-calling surface, not
@@ -447,21 +458,48 @@ export class ChatCommand extends BaseCommand {
447
458
  }
448
459
  const parsed = parseRequestSync(message);
449
460
  const dispatchDecision = resolvePipelineDispatch(parsed, { dev: opts.dev, text: message });
450
- const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation, systemPolicy: opts.systemPolicy });
461
+ const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, onFinding: opts.onFinding, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation, systemPolicy: opts.systemPolicy, debugSurface: opts.debugSurface ?? 'cli-chat', debugSession: opts.debugSession, worktree: opts.worktree, keepWorktree: opts.keepWorktree, resume: opts.resume });
451
462
  // No-model fallback: the tool loop could not generate a single response
452
463
  // AND the rules assessed a high-confidence pipeline intent — run the
453
464
  // pipeline directly (rules decide only when the model is unavailable; the
454
465
  // pipeline resolves its own working provider/model).
455
- if (answer.generationFailed && dispatchDecision.dispatch && !dispatchDecision.needConfirm) {
466
+ // WS5 (#27) — `!answer.refused` is not an optimisation, it is the guard: a
467
+ // turn that REFUSED to run (see `runChatAnswer`) is failed, but it is not
468
+ // UNANSWERED, and the no-model fallback exists for the second case. Re-
469
+ // dispatching it here would run the ask on another engine entirely — the one
470
+ // path that can run it UNISOLATED while the caller asked for isolation.
471
+ if (answer.generationFailed && !answer.refused && dispatchDecision.dispatch && !dispatchDecision.needConfirm) {
456
472
  const r = await runPipelineTool(message, this.configManager, { provider: type, model, board: false });
457
473
  // `success`, not `error`: a pipeline that RAN and failed reports its
458
474
  // outcome in `summary` and only sometimes sets `error`, so keying off
459
475
  // `error` alone returned a failed run's summary with NO failure flag —
460
476
  // i.e. reported it as a successful turn on every surface.
461
477
  if (!r.success) {
462
- return { content: '', followups: [], generationFailed: true, provider: type, model };
478
+ return {
479
+ content: '',
480
+ followups: [],
481
+ generationFailed: true,
482
+ provider: type,
483
+ model,
484
+ transport: 'none',
485
+ // WS5 — the isolation/resume this turn had, carried even on this path:
486
+ // the worktree was made and measured before the fallback ran, and a
487
+ // caller that never hears about it cannot tell an isolated turn from one
488
+ // that ran in the real tree.
489
+ ...this.turnEnvelopeOf(answer),
490
+ };
463
491
  }
464
- return { content: r.result?.summary ?? '', followups: [], provider: type, model };
492
+ // A pipeline turn carries no tool transport at all — reported as `none`
493
+ // rather than left silent, so a caller can tell "no transport" apart from
494
+ // "this surface never said".
495
+ return {
496
+ content: r.result?.summary ?? '',
497
+ followups: [],
498
+ provider: type,
499
+ model,
500
+ transport: 'none',
501
+ ...this.turnEnvelopeOf(answer),
502
+ };
465
503
  }
466
504
  // E3b: strip raw suggest_followups JSON embedded in content by the model
467
505
  const cleanContent = stripToolCallArtifacts(answer.content || '');
@@ -475,8 +513,37 @@ export class ChatCommand extends BaseCommand {
475
513
  unverifiedActionClaim: answer.unverifiedActionClaim,
476
514
  unfulfilledPromise: answer.unfulfilledPromise,
477
515
  undeliveredArtifact: answer.undeliveredArtifact,
516
+ // WS1 — the findings this turn recorded (empty when it recorded none).
517
+ findings: answer.findings ?? [],
478
518
  provider: type,
479
519
  model,
520
+ // R2 — the transport the loop's model-call seam reported for this turn.
521
+ transport: answer.transport,
522
+ // WS5 — and the isolation/resume the turn had. The engine reports them on
523
+ // ITS result; this method builds a new object, so without this spread they
524
+ // were dropped at the boundary — measured as a surface that isolated its
525
+ // turn correctly and then told its caller nothing about it.
526
+ ...this.turnEnvelopeOf(answer),
527
+ // WS5 — and whether the turn refused to run at all, for the same reason:
528
+ // a caller that cannot tell a refusal from a failed generation retries it,
529
+ // and there is nothing to retry (see the refusal return in `runChatAnswer`).
530
+ ...(answer.refused ? { refused: true } : {}),
531
+ };
532
+ }
533
+ /**
534
+ * WS5 (#27) — the isolation/resume facts of a finished turn, in the shape this
535
+ * command's callers read.
536
+ *
537
+ * Extracted because THREE returns in `answerOnce` hand back a turn the engine
538
+ * produced (the final answer, and the two no-model pipeline fallbacks), and a
539
+ * fact that travelled on only one of them would be a capability that disappears
540
+ * exactly when the model was unavailable — which is a failure mode, not an edge
541
+ * case.
542
+ */
543
+ turnEnvelopeOf(answer) {
544
+ return {
545
+ ...(answer.worktree ? { worktree: answer.worktree } : {}),
546
+ ...(answer.resume ? { resume: answer.resume } : {}),
480
547
  };
481
548
  }
482
549
  create() {
@@ -488,11 +555,41 @@ export class ChatCommand extends BaseCommand {
488
555
  .option('-m, --model <model>', 'Model to use (if omitted, an interactive picker will appear)')
489
556
  .option('--no-cache', 'Disable response caching')
490
557
  .option('-d, --dev', 'Always dispatch requests to the coding pipeline (no confirmation)', false)
558
+ // WS5 (#27) — isolation and partial resume, as the two things an operator
559
+ // asks for by hand. Both default to OFF and both are also readable from the
560
+ // environment (`NUVIRA_ISOLATE` / `NUVIRA_RESUME`), which is how the
561
+ // surfaces with no command line ask.
562
+ // NO `false` DEFAULT on any of the three, and that is load-bearing: commander
563
+ // would then hand this command `worktree: false` for a flag the operator never
564
+ // typed, an explicit FALSE outranks the environment in
565
+ // `resolveIsolationRequest`, and `NUVIRA_ISOLATE=1` would be silently ignored
566
+ // on the one surface whose flags outrank it. Absent is `undefined` — "nobody
567
+ // said" — which is what lets the environment ask for these on the CLI too.
568
+ .option('--worktree', 'Run this turn in its own git worktree of the project and report the diff against the base commit. Refuses rather than running unisolated when the directory cannot be isolated (also asked for by NUVIRA_ISOLATE=1)')
569
+ .option('--keep-worktree', 'Keep the isolated worktree after the turn instead of removing it')
570
+ .option('--resume [id]', 'Replay the recorded steps of this ask whose input is unchanged instead of paying for them again (defaults to the record for this goal + directory; also asked for by NUVIRA_RESUME=1)')
491
571
  .action(async (prompt, options) => {
492
572
  await this.execute(prompt, options || {});
493
573
  });
494
574
  return command;
495
575
  }
576
+ /**
577
+ * WS5 (#27) — the isolation/resume request the CLI's own flags carry.
578
+ *
579
+ * One place, because three call sites in this command hand the request to the
580
+ * shared engine (the one-shot turn, its picked followups, and every REPL
581
+ * message) and a request that reached only some of them would be a flag that
582
+ * worked until the second message. `undefined` (no flag) is passed through as
583
+ * `undefined` rather than `false`, which is what lets the environment ask for
584
+ * isolation on a surface the CLI did not.
585
+ */
586
+ ws5Overrides(options) {
587
+ return {
588
+ worktree: options?.worktree,
589
+ keepWorktree: options?.keepWorktree,
590
+ resume: options?.resume,
591
+ };
592
+ }
496
593
  async execute(prompt, options) {
497
594
  // Apply the active model state from `nuvira model switch` as defaults
498
595
  const activeOpts = applyActiveModel({ provider: options?.provider, model: options?.model });
@@ -536,7 +633,7 @@ export class ChatCommand extends BaseCommand {
536
633
  const available = await provider.isAvailable();
537
634
  if (!available) {
538
635
  logger.error(`${provider.name} is not available. Check your configuration.`);
539
- logger.info(`Run: agent-baba-d config --help`);
636
+ logger.info(`Run: agent-nuvira config --help`);
540
637
  return;
541
638
  }
542
639
  // ── Setup SIGINT (Ctrl+C) handler for graceful exit ──────────────
@@ -583,12 +680,17 @@ export class ChatCommand extends BaseCommand {
583
680
  // failed entirely), never as a bypass.
584
681
  const parsed = parseRequestSync(prompt);
585
682
  const dispatchDecision = resolvePipelineDispatch(parsed, { dev: options?.dev, text: prompt });
586
- const answer = await this.runChatAnswer(prompt, [], { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parsed);
683
+ const answer = await this.runChatAnswer(prompt, [], { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parsed,
684
+ // WS5 (#27) — isolation and resume ride on the CLI's own flags. Passed
685
+ // per turn: in the REPL each message is its own turn (see the manual).
686
+ this.ws5Overrides(options));
587
687
  // No-model fallback: the tool loop could not generate a single response
588
688
  // AND the rules assessed a high-confidence pipeline intent — run the
589
689
  // pipeline directly (rules decide only when the model is unavailable;
590
690
  // the pipeline resolves its own working provider/model).
591
- if (answer.generationFailed && dispatchDecision.dispatch && !dispatchDecision.needConfirm) {
691
+ // WS5 — never on a REFUSED turn: the pipeline would run it in the real tree
692
+ // (see the guard in `answerOnce`).
693
+ if (answer.generationFailed && !answer.refused && dispatchDecision.dispatch && !dispatchDecision.needConfirm) {
592
694
  await runDeveloperMode(prompt, this.configManager, { provider: type, model });
593
695
  // After pipeline execution, show followups and continue conversation
594
696
  // (don't just return — keep user engaged with next steps)
@@ -628,7 +730,7 @@ export class ChatCommand extends BaseCommand {
628
730
  break;
629
731
  const next = await this.runChatAnswer(picked, history, { type, provider, model }, options || {}, cacheEnabled, { auto: autoMode }, parseRequestSync(picked),
630
732
  // P5 — a picked followup continues the previous execution.
631
- { continuation: true });
733
+ { continuation: true, ...this.ws5Overrides(options) });
632
734
  const nextText = stripToolCallArtifacts(next.content);
633
735
  if (nextText) {
634
736
  console.log('\n' + nextText + '\n');
@@ -719,11 +821,15 @@ export class ChatCommand extends BaseCommand {
719
821
  const answer = await withLogCorrelation({ sessionId: chatSessionId }, () => recordMetricTime('llm.answer.ms', () => this.runChatAnswer(message, history, session, options || {}, cacheEnabled, { auto: autoMode }, parsed,
720
822
  // P5 — a picked followup (or a typed one that matches the last
721
823
  // suggestions) is a continuation, not a fresh independent request.
722
- { continuation: pickedFollowup || isSuggestedFollowup(message, lastFollowups) })));
824
+ {
825
+ continuation: pickedFollowup || isSuggestedFollowup(message, lastFollowups),
826
+ ...this.ws5Overrides(options),
827
+ })));
723
828
  // No-model fallback: the tool loop could not generate a single response
724
829
  // AND the rules assessed a high-confidence pipeline intent — run the
725
830
  // pipeline directly (rules decide only when the model is unavailable).
726
- if (answer.generationFailed && dispatchDecision.dispatch && !dispatchDecision.needConfirm) {
831
+ // WS5 — never on a REFUSED turn, for the same reason as above.
832
+ if (answer.generationFailed && !answer.refused && dispatchDecision.dispatch && !dispatchDecision.needConfirm) {
727
833
  await runDeveloperMode(message, this.configManager, { provider: type, model });
728
834
  // After pipeline execution, continue conversation (don't just ask "press Enter")
729
835
  // The user can keep chatting or type /exit
@@ -819,6 +925,24 @@ export class ChatCommand extends BaseCommand {
819
925
  * interactive mode turns a chosen followup into the next message).
820
926
  */
821
927
  async runChatAnswer(message, history, session, options, cacheEnabled, mode, parsed, ctxOverrides) {
928
+ // WS2 (#24) — the optional session debug log for this turn. Null unless
929
+ // `NUVIRA_DEBUG_LOG` is set, so the off path is one boolean check; when on,
930
+ // the events below are redacted and bounded, and the file is written at the
931
+ // END so its header can name the backend that actually served the turn.
932
+ const debugLog = sessionDebugLog({
933
+ surface: ctxOverrides?.debugSurface ?? 'cli-chat',
934
+ goal: message,
935
+ ...(ctxOverrides?.debugSession ? { session: ctxOverrides.debugSession } : {}),
936
+ backend: { engine: 'loop', provider: session.type, ...(session.model ? { model: session.model } : {}) },
937
+ });
938
+ debugLog?.event('turn.start', { provider: session.type });
939
+ // WS3 (#25) — the turn's span root, when span export is on (else null). Same
940
+ // identity as the log's: one surface, one conversation, one turn.
941
+ const otelSpan = await startTurnSpan({
942
+ surface: ctxOverrides?.debugSurface ?? 'cli-chat',
943
+ ...(ctxOverrides?.debugSession ? { session: ctxOverrides.debugSession } : {}),
944
+ goal: message,
945
+ });
822
946
  // Cache check first (same as the legacy path).
823
947
  const cache = getCache();
824
948
  const cacheModel = this.cacheModelFor(session);
@@ -833,6 +957,12 @@ export class ChatCommand extends BaseCommand {
833
957
  history.push({ role: 'user', content: message });
834
958
  history.push({ role: 'assistant', content: cachedResult });
835
959
  this.memoryNoteTurn(message, cachedResult);
960
+ // WS2 — a cache replay reached no model, so the log says exactly that
961
+ // rather than borrowing an attribution from a turn that did not run.
962
+ debugLog?.event('cache.hit', { chars: cachedResult.length });
963
+ const cacheNotice = debugLogNotice(ctxOverrides?.debugSurface ?? 'cli-chat', debugLog?.write() ?? null);
964
+ if (cacheNotice)
965
+ logger.info(cacheNotice);
836
966
  return { content: cachedResult };
837
967
  }
838
968
  }
@@ -840,6 +970,107 @@ export class ChatCommand extends BaseCommand {
840
970
  // Cache must never break the turn.
841
971
  }
842
972
  }
973
+ // ─── WS5 (#27) — ISOLATION AND RESUME, the whole turn's envelope ────────
974
+ //
975
+ // HERE, and not in each caller, because this is the seam every in-process
976
+ // surface already shares: the CLI's interactive REPL and one-shot answer, the
977
+ // dashboard console, the gateway's inbound chat and `execute`'s direct-answer
978
+ // arm all reach the tool loop through this method. A wrapper in each caller
979
+ // would be four copies of one policy, and the one that drifted would be the
980
+ // surface that quietly ran in the real tree.
981
+ //
982
+ // It sits AFTER the response-cache check on purpose: a cached answer does no
983
+ // work, so there is nothing to isolate and no step to replay — paying for a
984
+ // git checkout to replay a cached string would be pure cost.
985
+ /**
986
+ * Where a WS5 notice goes: the surface's own progress channel when it has one
987
+ * (the dashboard renders it, the gateway logs it), else this process's log.
988
+ *
989
+ * Deliberately NOT `ctxOverrides.onProgress` directly: the CLI passes no
990
+ * progress sink for a one-shot turn, and a notice that reached nothing would
991
+ * hide exactly the facts it exists for — which directory the turn is isolated
992
+ * in, and what it changed.
993
+ */
994
+ const report = ctxOverrides?.onProgress ?? ((line) => void logger.info(line));
995
+ const projectDir = ctxOverrides?.projectPath || process.cwd();
996
+ const isolationRequest = resolveIsolationRequest({
997
+ worktree: ctxOverrides?.worktree,
998
+ keepWorktree: ctxOverrides?.keepWorktree,
999
+ });
1000
+ const isolation = beginIsolation({ request: isolationRequest, repoCwd: projectDir, label: message });
1001
+ if (isolation && !isolation.ok) {
1002
+ // REFUSED, not degraded: the operator asked for isolation on purpose, and a
1003
+ // turn that ran unisolated while its result said otherwise would be the one
1004
+ // outcome this capability exists to prevent. Reported as a FAILED turn, so
1005
+ // no surface presents it as an answer.
1006
+ report(isolation.refusal);
1007
+ return {
1008
+ content: isolation.refusal,
1009
+ followups: [],
1010
+ // BOTH flags, and they say different things. `generationFailed` keeps the
1011
+ // turn a FAILED one, so no surface renders the refusal as an answer.
1012
+ // `refused` says WHY it failed — the turn never ran, no model was called —
1013
+ // and that distinction is load-bearing: the caller's no-model fallback
1014
+ // keys off `generationFailed` alone, so without this a refused turn was
1015
+ // silently re-dispatched to the PIPELINE, which ran the ask in the real
1016
+ // tree. Measured: `nuvira chat "write a file…" --worktree` outside a git
1017
+ // repository printed a three-task pipeline board and never mentioned the
1018
+ // refusal — the exact outcome isolation exists to prevent.
1019
+ generationFailed: true,
1020
+ refused: true,
1021
+ };
1022
+ }
1023
+ const worktree = isolation?.ok ? isolation.worktree : null;
1024
+ /**
1025
+ * The directory this turn works in: the worktree when isolated, the attached
1026
+ * project when one was given, else the process's own cwd.
1027
+ *
1028
+ * EVERY path that resolves a directory from here on reads this — the tools'
1029
+ * `cwd`, the ambient project snapshot, the working-state ledger — because a
1030
+ * turn that is isolated for its tools but reads its context from the original
1031
+ * tree is not isolated, it is confused.
1032
+ */
1033
+ const turnCwd = worktree?.dir ?? projectDir;
1034
+ // The ledger is only opened when a resume was asked for (a `--resume`, or the
1035
+ // environment asking for every turn on this surface). An ordinary turn never
1036
+ // touches the record store: no read, no write, no directory created.
1037
+ const resumeRequest = resolveResumeRequest({ resume: ctxOverrides?.resume });
1038
+ const resume = resumeRequest
1039
+ ? openResume({ goal: message, cwd: turnCwd, resume: resumeRequest })
1040
+ : null;
1041
+ if (worktree)
1042
+ report(worktreeNotice(worktree));
1043
+ // WS5 — what the RECORD holds, said before the turn. The outcome (what was
1044
+ // replayed, what it cost) is reported in `finish` below, where it is knowable —
1045
+ // this line used to state the outcome here, which meant every resumed turn
1046
+ // announced "nothing to replay" before it had tried anything.
1047
+ if (resume)
1048
+ report(resume.ledger.openNotice());
1049
+ /**
1050
+ * Attach this turn's isolation and resume outcomes to whatever it returns.
1051
+ *
1052
+ * A helper at every return rather than a `finally`, because the outcomes have
1053
+ * to ride ON the result a caller is waiting for: a diff reported later (or by
1054
+ * a separate command) is a diff most callers never see, and `endIsolation`
1055
+ * never throws, so a cleanup failure cannot replace the turn's own answer
1056
+ * with a git error.
1057
+ */
1058
+ const finish = (result) => {
1059
+ const extra = {};
1060
+ if (worktree) {
1061
+ const outcome = endIsolation(worktree, { keep: isolationRequest.keep });
1062
+ extra.worktree = outcome;
1063
+ report(outcome.notice);
1064
+ }
1065
+ if (resume) {
1066
+ const outcome = closeResume(resume, { goal: message, cwd: turnCwd });
1067
+ extra.resume = outcome;
1068
+ // The wording lives in `ResumeOutcome.notice` — one sentence, every
1069
+ // surface, including the reason when nothing replayed.
1070
+ report(outcome.notice);
1071
+ }
1072
+ return { ...result, ...extra };
1073
+ };
843
1074
  history.push({ role: 'user', content: message });
844
1075
  // System prompt: base identity + the tool contract — the
845
1076
  // model clarifies with ask_user and ends every response with followups.
@@ -895,7 +1126,7 @@ export class ChatCommand extends BaseCommand {
895
1126
  let ambientProjectContext;
896
1127
  if (!ctxOverrides?.projectContext) {
897
1128
  try {
898
- const built = await buildLoopProjectContext(ctxOverrides?.projectPath || process.cwd());
1129
+ const built = await buildLoopProjectContext(turnCwd);
899
1130
  if (built)
900
1131
  ambientProjectContext = built;
901
1132
  }
@@ -908,7 +1139,7 @@ export class ChatCommand extends BaseCommand {
908
1139
  // what previous turns already established. This is the fix for the
909
1140
  // calculator session's core drift (it re-diagnosed the same root cause six
910
1141
  // times, then undid its own earlier fixes).
911
- const workingStatePath = ctxOverrides?.projectPath || process.cwd();
1142
+ const workingStatePath = turnCwd;
912
1143
  const workingStateBlock = formatWorkingState(getWorkingState(workingStatePath));
913
1144
  // Session 3 — channel/format policy lives in the STABLE layer. It is
914
1145
  // identical on every message, so keeping it here makes the system prompt
@@ -961,6 +1192,15 @@ export class ChatCommand extends BaseCommand {
961
1192
  // G3 — the files this turn mutates, observed on the tool event stream so
962
1193
  // the ledger can remember them (the loop reports tool NAMES, not paths).
963
1194
  const touchedFiles = new Set();
1195
+ /**
1196
+ * WS1 — the findings this turn RECORDED, in call order.
1197
+ *
1198
+ * Collected from the same `finding:recorded` event the GUI hears, so the
1199
+ * result the caller returns and the live card a user watches are fed by one
1200
+ * source rather than two that can disagree. Empty is a real answer ("this
1201
+ * surface reports findings, and this turn recorded none"), never silence.
1202
+ */
1203
+ const findings = [];
964
1204
  /**
965
1205
  * G18 — the turn's trace id, for the autonomy-gate events that tools emit
966
1206
  * DURING the loop. Assigned a few lines below (the trace begins once the
@@ -973,7 +1213,7 @@ export class ChatCommand extends BaseCommand {
973
1213
  loadedExtraTools,
974
1214
  // P4 — when a project is attached, scope tools to its root so the
975
1215
  // agent operates inside the project (not the dashboard server's cwd).
976
- cwd: ctxOverrides?.projectPath || process.cwd(),
1216
+ cwd: turnCwd,
977
1217
  emit: (event, data, source) => {
978
1218
  // G3 — collect mutated file paths from `tool:started` (which carries
979
1219
  // the arguments) so the working-state ledger knows what changed.
@@ -991,6 +1231,19 @@ export class ChatCommand extends BaseCommand {
991
1231
  if (ctxOverrides?.onToolCall && (event === 'tool:started' || event === 'tool:called')) {
992
1232
  ctxOverrides.onToolCall(event === 'tool:started' ? 'started' : 'called', data);
993
1233
  }
1234
+ // WS2 — the same lifecycle into the session debug log: a bug report
1235
+ // needs the tool NAMES and their outcomes, in order. `write()` is never
1236
+ // called here (the log is buffered and written once at turn end); this
1237
+ // records, it does not persist per call.
1238
+ if (debugLog && (event === 'tool:started' || event === 'tool:called')) {
1239
+ const call = data;
1240
+ if (call?.tool) {
1241
+ debugLog.event(event === 'tool:started' ? 'tool.start' : 'tool.end', {
1242
+ tool: call.tool,
1243
+ ...(call.ok === undefined ? {} : { ok: call.ok }),
1244
+ });
1245
+ }
1246
+ }
994
1247
  // P0.7 — forward plan mutations to the GUI (structured checklist).
995
1248
  if (ctxOverrides?.onPlanChange && event === 'plan:changed') {
996
1249
  ctxOverrides.onPlanChange(data);
@@ -1004,6 +1257,23 @@ export class ChatCommand extends BaseCommand {
1004
1257
  if (ctxOverrides?.onSkillDraft && event === 'skill:draft') {
1005
1258
  ctxOverrides.onSkillDraft(data);
1006
1259
  }
1260
+ // WS1 — a finding the turn recorded. Kept on this surface's own result
1261
+ // (so a caller that never sees the GUI still gets the verdict) AND
1262
+ // forwarded to the caller's live view, one event, two readers.
1263
+ if (event === FINDING_EVENT) {
1264
+ const finding = data;
1265
+ findings.push(finding);
1266
+ ctxOverrides?.onFinding?.(finding);
1267
+ // WS2 — a verdict is exactly the kind of fact a bug report is missing.
1268
+ debugLog?.event('finding', `${finding.verdict} ${finding.claim}`);
1269
+ // WS3 — and a span EVENT rather than a span: a finding has no
1270
+ // duration, so a point-in-time fact is the honest shape for it.
1271
+ otelSpan?.event('nuvira.finding', {
1272
+ 'nuvira.verdict': finding.verdict,
1273
+ 'nuvira.claim': finding.claim,
1274
+ 'nuvira.outcome': finding.outcome,
1275
+ });
1276
+ }
1007
1277
  // G18 — an autonomy gate DECIDING to proceed is a fact about the turn
1008
1278
  // ("this change was applied without asking, and here is why"), not just
1009
1279
  // a bus notification: record it on the turn's trace so the decision is
@@ -1150,6 +1420,13 @@ export class ChatCommand extends BaseCommand {
1150
1420
  // and is now additionally gated on the model having the context for it.
1151
1421
  toolExposure: harness.exposure,
1152
1422
  maxParallelReads: harness.maxParallelReads,
1423
+ // WS3 — the turn span the loop hangs each tool call under.
1424
+ otel: otelSpan,
1425
+ // WS4 — the label a tool hook reports this call under.
1426
+ surface: ctxOverrides?.debugSurface ?? 'cli-chat',
1427
+ // WS5 — the resume ledger, when this turn was asked to resume. Omitted
1428
+ // entirely otherwise, so an ordinary turn never consults it.
1429
+ ...(resume ? { resume: resume.ledger } : {}),
1153
1430
  onToken: ctxOverrides?.onToken,
1154
1431
  signal: ctxOverrides?.signal,
1155
1432
  // G18 — the same sink the execute loop uses: tool calls, gate decisions
@@ -1220,6 +1497,12 @@ export class ChatCommand extends BaseCommand {
1220
1497
  unverifiedEditClaim: result.unverifiedEditClaim,
1221
1498
  undeliveredArtifact: result.undeliveredArtifact,
1222
1499
  }));
1500
+ // WS1 (#23) — persist the turn's findings (claim, outcome, evidence and the
1501
+ // gate's verdict) on the trace, so the verdicts can be audited from the
1502
+ // Trace tab after the run instead of only existing in this turn's return.
1503
+ // Best-effort by construction; `endTrace` above cleared the in-progress id,
1504
+ // so the trace id is passed explicitly.
1505
+ recordTraceFindings(chatTraceId, findings);
1223
1506
  // G3 — record what this turn actually did so the NEXT turn starts from it
1224
1507
  // (files changed, whether anything verified the work, and whether the user
1225
1508
  // reported a regression). Best-effort: the ledger must never break a turn.
@@ -1301,7 +1584,50 @@ export class ChatCommand extends BaseCommand {
1301
1584
  // directly, so a model that wrote the tool JSON as text used to leak it
1302
1585
  // into the chat. The loop already salvages such blocks into real tool
1303
1586
  // calls; this is the belt-and-braces strip for any residue.
1304
- return {
1587
+ // WS2 — close the session debug log. The header names the backend that
1588
+ // ACTUALLY served the turn (`lastAttempt`, updated by the provider walk),
1589
+ // not the pair this surface merely resolved before the turn started — those
1590
+ // diverge exactly when failover happens, which is when a bug report needs
1591
+ // the right answer. Best-effort: a log that cannot be written must never
1592
+ // affect the answer.
1593
+ if (debugLog) {
1594
+ const servedModel = this.lastAttempt?.model ?? session.model;
1595
+ debugLog.backendOf({
1596
+ provider: this.lastAttempt?.provider ?? session.type,
1597
+ ...(servedModel ? { model: servedModel } : {}),
1598
+ transport: result.transport ?? null,
1599
+ });
1600
+ debugLog.event('turn.end', {
1601
+ generationFailed: result.generationFailed === true,
1602
+ cancelled: result.cancelled === true,
1603
+ bounded: result.bounded === true,
1604
+ contentChars: result.content.length,
1605
+ toolCalls: result.toolCalls?.length ?? 0,
1606
+ findings: findings.length,
1607
+ });
1608
+ const notice = debugLogNotice(ctxOverrides?.debugSurface ?? 'cli-chat', debugLog.write());
1609
+ if (notice)
1610
+ logger.info(notice);
1611
+ }
1612
+ // WS3 (#25) — close the turn span and ship it. The status is the turn's own
1613
+ // outcome, so a failed turn is a RED span in the collector rather than an
1614
+ // absent one — the same rule the debug log follows for a crash.
1615
+ if (otelSpan) {
1616
+ otelSpan.attr('nuvira.findings', findings.length);
1617
+ otelSpan.end({
1618
+ ok: result.generationFailed !== true && result.cancelled !== true,
1619
+ ...(result.generationFailed === true
1620
+ ? { message: 'the turn did not produce a usable answer' }
1621
+ : result.cancelled === true
1622
+ ? { message: 'the turn was cancelled' }
1623
+ : {}),
1624
+ });
1625
+ const otelLine = otelNoticeOnce(ctxOverrides?.debugSurface ?? 'cli-chat');
1626
+ if (otelLine)
1627
+ logger.info(otelLine);
1628
+ await flushSpans();
1629
+ }
1630
+ return finish({
1305
1631
  content: stripToolCallArtifacts(result.content),
1306
1632
  generationFailed: result.generationFailed,
1307
1633
  cancelled: result.cancelled,
@@ -1311,7 +1637,11 @@ export class ChatCommand extends BaseCommand {
1311
1637
  unverifiedActionClaim: result.unverifiedActionClaim,
1312
1638
  unfulfilledPromise: result.unfulfilledPromise,
1313
1639
  undeliveredArtifact: result.undeliveredArtifact,
1314
- };
1640
+ // R2 — the transport this turn travelled on (interactive REPL path).
1641
+ transport: result.transport,
1642
+ // WS1 — the findings this turn recorded, with their verdicts.
1643
+ findings,
1644
+ });
1315
1645
  }
1316
1646
  /**
1317
1647
  * E3b — the model-call step for the tool loop:
@@ -1473,13 +1803,13 @@ export class ChatCommand extends BaseCommand {
1473
1803
  if (sink && typeof prov.generateToolsStream === 'function') {
1474
1804
  const result = await prov.generateToolsStream(messages, schemas, { ...options, model: effectiveModel, signal: abort }, sink);
1475
1805
  confuseCheck(result.content, result.toolCalls.length > 0);
1476
- return answered(result);
1806
+ return answered({ ...result, transport: 'native' });
1477
1807
  }
1478
1808
  const result = await prov.generateTools(messages, schemas, { ...options, model: effectiveModel, signal: abort });
1479
1809
  confuseCheck(result.content, result.toolCalls.length > 0);
1480
1810
  if (sink && result.content)
1481
1811
  sink(result.content);
1482
- return answered(result);
1812
+ return answered({ ...result, transport: 'native' });
1483
1813
  }
1484
1814
  catch (err) {
1485
1815
  // S3: a tool-call 400 often carries the model's COMPLETE answer in
@@ -1498,7 +1828,9 @@ export class ChatCommand extends BaseCommand {
1498
1828
  const toolCalls = salvaged.followups?.length
1499
1829
  ? [{ id: 'call_salvage_1', name: 'suggest_followups', arguments: { followups: salvaged.followups } }]
1500
1830
  : [];
1501
- return answered({ content: salvaged.content, toolCalls });
1831
+ // R2 — the answer was excavated from a NATIVE tool-call attempt's
1832
+ // 400 payload, so it is still the native transport's output.
1833
+ return answered({ content: salvaged.content, toolCalls, transport: 'native' });
1502
1834
  }
1503
1835
  // The MODEL itself cannot do native tool calling — Groq answers
1504
1836
  // 400 "`tool calling` is not supported with this model". That is
@@ -1532,7 +1864,9 @@ export class ChatCommand extends BaseCommand {
1532
1864
  }
1533
1865
  const { text, calls } = extractFallbackToolCalls(raw);
1534
1866
  confuseCheck(text, calls.length > 0);
1535
- return answered({ content: text, toolCalls: calls });
1867
+ // R2 — this IS the fallback transport, so say so: the same fact the
1868
+ // subagent child announces, and what makes a chat turn comparable with it.
1869
+ return answered({ content: text, toolCalls: calls, transport: 'json' });
1536
1870
  };
1537
1871
  try {
1538
1872
  // Same-provider transient retry FIRST (see the helper's contract): a