agent-nuvira 3.3.0 → 3.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (231) hide show
  1. package/.agents/skills/code-assessment/SKILL.md +10 -1
  2. package/.agents/skills/index.json +1 -1
  3. package/README.md +47 -9
  4. package/dist/agents/agents/runner.d.ts +9 -0
  5. package/dist/agents/agents/runner.d.ts.map +1 -1
  6. package/dist/agents/agents/runner.js +34 -12
  7. package/dist/agents/agents/runner.js.map +1 -1
  8. package/dist/agents/agents/writer-tool-calling.d.ts.map +1 -1
  9. package/dist/agents/agents/writer-tool-calling.js +7 -2
  10. package/dist/agents/agents/writer-tool-calling.js.map +1 -1
  11. package/dist/agents/agents/writer.d.ts.map +1 -1
  12. package/dist/agents/agents/writer.js +7 -1
  13. package/dist/agents/agents/writer.js.map +1 -1
  14. package/dist/agents/artifact-verification.d.ts +102 -0
  15. package/dist/agents/artifact-verification.d.ts.map +1 -0
  16. package/dist/agents/artifact-verification.js +216 -0
  17. package/dist/agents/artifact-verification.js.map +1 -0
  18. package/dist/agents/checkpoint-store.d.ts +73 -0
  19. package/dist/agents/checkpoint-store.d.ts.map +1 -1
  20. package/dist/agents/checkpoint-store.js +161 -0
  21. package/dist/agents/checkpoint-store.js.map +1 -1
  22. package/dist/agents/credential-store.d.ts +120 -0
  23. package/dist/agents/credential-store.d.ts.map +1 -1
  24. package/dist/agents/credential-store.js +358 -19
  25. package/dist/agents/credential-store.js.map +1 -1
  26. package/dist/agents/orchestrator.d.ts +10 -1
  27. package/dist/agents/orchestrator.d.ts.map +1 -1
  28. package/dist/agents/orchestrator.js +234 -75
  29. package/dist/agents/orchestrator.js.map +1 -1
  30. package/dist/agents/phase-engine.d.ts +61 -0
  31. package/dist/agents/phase-engine.d.ts.map +1 -1
  32. package/dist/agents/phase-engine.js +68 -2
  33. package/dist/agents/phase-engine.js.map +1 -1
  34. package/dist/agents/prompt-assembly.d.ts +11 -0
  35. package/dist/agents/prompt-assembly.d.ts.map +1 -1
  36. package/dist/agents/prompt-assembly.js +17 -0
  37. package/dist/agents/prompt-assembly.js.map +1 -1
  38. package/dist/agents/release-preflight.d.ts +121 -0
  39. package/dist/agents/release-preflight.d.ts.map +1 -0
  40. package/dist/agents/release-preflight.js +259 -0
  41. package/dist/agents/release-preflight.js.map +1 -0
  42. package/dist/agents/release-runner.d.ts +157 -0
  43. package/dist/agents/release-runner.d.ts.map +1 -0
  44. package/dist/agents/release-runner.js +719 -0
  45. package/dist/agents/release-runner.js.map +1 -0
  46. package/dist/agents/step-handoff.d.ts +190 -0
  47. package/dist/agents/step-handoff.d.ts.map +1 -0
  48. package/dist/agents/step-handoff.js +443 -0
  49. package/dist/agents/step-handoff.js.map +1 -0
  50. package/dist/cli/benchmark.d.ts.map +1 -1
  51. package/dist/cli/benchmark.js +11 -4
  52. package/dist/cli/benchmark.js.map +1 -1
  53. package/dist/cli/chat.d.ts.map +1 -1
  54. package/dist/cli/chat.js +25 -4
  55. package/dist/cli/chat.js.map +1 -1
  56. package/dist/cli/cli-program.d.ts.map +1 -1
  57. package/dist/cli/cli-program.js +9 -2
  58. package/dist/cli/cli-program.js.map +1 -1
  59. package/dist/cli/commands.d.ts +2 -1
  60. package/dist/cli/commands.d.ts.map +1 -1
  61. package/dist/cli/commands.js +3 -2
  62. package/dist/cli/commands.js.map +1 -1
  63. package/dist/cli/config.js +2 -2
  64. package/dist/cli/config.js.map +1 -1
  65. package/dist/cli/credentials.d.ts +28 -0
  66. package/dist/cli/credentials.d.ts.map +1 -0
  67. package/dist/cli/credentials.js +213 -0
  68. package/dist/cli/credentials.js.map +1 -0
  69. package/dist/cli/dashboard.d.ts +26 -0
  70. package/dist/cli/dashboard.d.ts.map +1 -1
  71. package/dist/cli/dashboard.js +123 -3
  72. package/dist/cli/dashboard.js.map +1 -1
  73. package/dist/cli/failover-runner.d.ts.map +1 -1
  74. package/dist/cli/failover-runner.js +9 -3
  75. package/dist/cli/failover-runner.js.map +1 -1
  76. package/dist/cli/intent.d.ts +1 -1
  77. package/dist/cli/intent.js +2 -2
  78. package/dist/cli/intent.js.map +1 -1
  79. package/dist/cli/loop-executor.d.ts +7 -0
  80. package/dist/cli/loop-executor.d.ts.map +1 -1
  81. package/dist/cli/loop-executor.js +75 -7
  82. package/dist/cli/loop-executor.js.map +1 -1
  83. package/dist/cli/nlu.d.ts.map +1 -1
  84. package/dist/cli/nlu.js +8 -2
  85. package/dist/cli/nlu.js.map +1 -1
  86. package/dist/cli/process-control.d.ts +15 -0
  87. package/dist/cli/process-control.d.ts.map +1 -1
  88. package/dist/cli/process-control.js +38 -14
  89. package/dist/cli/process-control.js.map +1 -1
  90. package/dist/cli/publish.d.ts +22 -0
  91. package/dist/cli/publish.d.ts.map +1 -1
  92. package/dist/cli/publish.js +109 -13
  93. package/dist/cli/publish.js.map +1 -1
  94. package/dist/config/live-credentials.d.ts +57 -0
  95. package/dist/config/live-credentials.d.ts.map +1 -0
  96. package/dist/config/live-credentials.js +126 -0
  97. package/dist/config/live-credentials.js.map +1 -0
  98. package/dist/config/paths.d.ts +18 -0
  99. package/dist/config/paths.d.ts.map +1 -1
  100. package/dist/config/paths.js +25 -0
  101. package/dist/config/paths.js.map +1 -1
  102. package/dist/federation/a2a-types.js +1 -1
  103. package/dist/federation/a2a-types.js.map +1 -1
  104. package/dist/forwarded.d.ts +2 -0
  105. package/dist/forwarded.d.ts.map +1 -0
  106. package/dist/forwarded.js +2 -0
  107. package/dist/forwarded.js.map +1 -0
  108. package/dist/inference/model-validator.d.ts +6 -1
  109. package/dist/inference/model-validator.d.ts.map +1 -1
  110. package/dist/inference/model-validator.js +7 -2
  111. package/dist/inference/model-validator.js.map +1 -1
  112. package/dist/inference/route-resolver.d.ts +116 -0
  113. package/dist/inference/route-resolver.d.ts.map +1 -0
  114. package/dist/inference/route-resolver.js +159 -0
  115. package/dist/inference/route-resolver.js.map +1 -0
  116. package/dist/inference/tool-call-utils.d.ts +44 -0
  117. package/dist/inference/tool-call-utils.d.ts.map +1 -1
  118. package/dist/inference/tool-call-utils.js +126 -4
  119. package/dist/inference/tool-call-utils.js.map +1 -1
  120. package/dist/learning/autonomy-policy.d.ts +24 -0
  121. package/dist/learning/autonomy-policy.d.ts.map +1 -1
  122. package/dist/learning/autonomy-policy.js +110 -23
  123. package/dist/learning/autonomy-policy.js.map +1 -1
  124. package/dist/learning/eval-framework.d.ts +24 -0
  125. package/dist/learning/eval-framework.d.ts.map +1 -1
  126. package/dist/learning/eval-framework.js +291 -0
  127. package/dist/learning/eval-framework.js.map +1 -1
  128. package/dist/learning/intent-envelope.d.ts +162 -0
  129. package/dist/learning/intent-envelope.d.ts.map +1 -0
  130. package/dist/learning/intent-envelope.js +235 -0
  131. package/dist/learning/intent-envelope.js.map +1 -0
  132. package/dist/learning/reasoning-trace.d.ts +2 -2
  133. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  134. package/dist/learning/reasoning-trace.js +25 -1
  135. package/dist/learning/reasoning-trace.js.map +1 -1
  136. package/dist/learning/resilient-call.d.ts.map +1 -1
  137. package/dist/learning/resilient-call.js +29 -17
  138. package/dist/learning/resilient-call.js.map +1 -1
  139. package/dist/learning/run-trace.d.ts +164 -0
  140. package/dist/learning/run-trace.d.ts.map +1 -0
  141. package/dist/learning/run-trace.js +342 -0
  142. package/dist/learning/run-trace.js.map +1 -0
  143. package/dist/learning/skill-store.d.ts.map +1 -1
  144. package/dist/learning/skill-store.js +35 -10
  145. package/dist/learning/skill-store.js.map +1 -1
  146. package/dist/learning/skill-types.d.ts +14 -0
  147. package/dist/learning/skill-types.d.ts.map +1 -1
  148. package/dist/learning/skill-types.js +19 -0
  149. package/dist/learning/skill-types.js.map +1 -1
  150. package/dist/mcp/catalog.js +2 -3
  151. package/dist/mcp/catalog.js.map +1 -1
  152. package/dist/mcp/manager.d.ts.map +1 -1
  153. package/dist/mcp/manager.js +5 -3
  154. package/dist/mcp/manager.js.map +1 -1
  155. package/dist/resources/command-manifest.json +10 -10
  156. package/dist/skills/bundled-skills.d.ts.map +1 -1
  157. package/dist/skills/bundled-skills.js +13 -2
  158. package/dist/skills/bundled-skills.js.map +1 -1
  159. package/dist/skills/secret-capture.d.ts.map +1 -1
  160. package/dist/skills/secret-capture.js +9 -4
  161. package/dist/skills/secret-capture.js.map +1 -1
  162. package/dist/tools/coding-tools.d.ts.map +1 -1
  163. package/dist/tools/coding-tools.js +33 -8
  164. package/dist/tools/coding-tools.js.map +1 -1
  165. package/dist/tools/credentials-tool.d.ts +24 -0
  166. package/dist/tools/credentials-tool.d.ts.map +1 -0
  167. package/dist/tools/credentials-tool.js +125 -0
  168. package/dist/tools/credentials-tool.js.map +1 -0
  169. package/dist/tools/edit-verification.d.ts +78 -0
  170. package/dist/tools/edit-verification.d.ts.map +1 -1
  171. package/dist/tools/edit-verification.js +213 -16
  172. package/dist/tools/edit-verification.js.map +1 -1
  173. package/dist/tools/git-tool.d.ts +25 -7
  174. package/dist/tools/git-tool.d.ts.map +1 -1
  175. package/dist/tools/git-tool.js +155 -15
  176. package/dist/tools/git-tool.js.map +1 -1
  177. package/dist/tools/loop-project-context.d.ts.map +1 -1
  178. package/dist/tools/loop-project-context.js +16 -0
  179. package/dist/tools/loop-project-context.js.map +1 -1
  180. package/dist/tools/loop-route-feed.d.ts +57 -0
  181. package/dist/tools/loop-route-feed.d.ts.map +1 -0
  182. package/dist/tools/loop-route-feed.js +101 -0
  183. package/dist/tools/loop-route-feed.js.map +1 -0
  184. package/dist/tools/pipeline-tool.d.ts.map +1 -1
  185. package/dist/tools/pipeline-tool.js +8 -2
  186. package/dist/tools/pipeline-tool.js.map +1 -1
  187. package/dist/tools/publish-tool.d.ts.map +1 -1
  188. package/dist/tools/publish-tool.js +76 -10
  189. package/dist/tools/publish-tool.js.map +1 -1
  190. package/dist/tools/registry.d.ts +52 -3
  191. package/dist/tools/registry.d.ts.map +1 -1
  192. package/dist/tools/registry.js +88 -11
  193. package/dist/tools/registry.js.map +1 -1
  194. package/dist/tools/run-cli.d.ts.map +1 -1
  195. package/dist/tools/run-cli.js +15 -2
  196. package/dist/tools/run-cli.js.map +1 -1
  197. package/dist/tools/run-terminal.d.ts +9 -0
  198. package/dist/tools/run-terminal.d.ts.map +1 -1
  199. package/dist/tools/run-terminal.js +103 -6
  200. package/dist/tools/run-terminal.js.map +1 -1
  201. package/dist/tools/skill-tool.d.ts.map +1 -1
  202. package/dist/tools/skill-tool.js +5 -0
  203. package/dist/tools/skill-tool.js.map +1 -1
  204. package/dist/tools/tool-loop.d.ts +40 -6
  205. package/dist/tools/tool-loop.d.ts.map +1 -1
  206. package/dist/tools/tool-loop.js +409 -14
  207. package/dist/tools/tool-loop.js.map +1 -1
  208. package/dist/tools/toolsets.js +2 -2
  209. package/dist/tools/toolsets.js.map +1 -1
  210. package/dist/web-dashboard/chat-console.d.ts +5 -0
  211. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  212. package/dist/web-dashboard/chat-console.js +32 -3
  213. package/dist/web-dashboard/chat-console.js.map +1 -1
  214. package/dist/web-dashboard/server.d.ts +46 -0
  215. package/dist/web-dashboard/server.d.ts.map +1 -1
  216. package/dist/web-dashboard/server.js +115 -1
  217. package/dist/web-dashboard/server.js.map +1 -1
  218. package/dist/web-dashboard/src/admin-auth.d.ts +49 -2
  219. package/dist/web-dashboard/src/admin-auth.d.ts.map +1 -1
  220. package/dist/web-dashboard/src/admin-auth.js +64 -3
  221. package/dist/web-dashboard/src/admin-auth.js.map +1 -1
  222. package/dist/web-dashboard/src/types.d.ts +41 -0
  223. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  224. package/dist/workflow/registry.js +1 -1
  225. package/dist/workflow/registry.js.map +1 -1
  226. package/package.json +9 -5
  227. package/src/web-dashboard/public/assets/index-CxDj7p6i.js +207 -0
  228. package/src/web-dashboard/public/assets/index-CxDj7p6i.js.map +1 -0
  229. package/src/web-dashboard/public/index.html +1 -1
  230. package/src/web-dashboard/public/assets/index-kCUkORm7.js +0 -207
  231. package/src/web-dashboard/public/assets/index-kCUkORm7.js.map +0 -1
@@ -16,14 +16,59 @@
16
16
  * blocks after its response text (contract in TOOL_CONTRACT_JSON).
17
17
  */
18
18
  import { getTool, toolJsonSchemas } from './registry.js';
19
- import { detectPermissionSeeking, replyAsksTheReader, requestAuthorizesWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
19
+ import { detectPermissionSeeking, isAffirmativeReply, replyAsksTheReader, requestAuthorizesWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
20
+ import { envelopeFromPlan, envelopeFromRequest, getEnvelope, grantEnvelope, isEnvelopeKey, } from '../learning/intent-envelope.js';
21
+ import { detectProcessComplaint, isTraceKey, repeatNudge, runTraceFor, RunTrace, } from '../learning/run-trace.js';
20
22
  import { wantsAuthoredArtifact } from '../learning/deliverable-class.js';
21
23
  import { normalizeFollowups } from './followup-utils.js';
22
- import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, VERIFICATION_NUDGE, AUTHORIZED_WORK_NUDGE, deliverableNudge, } from './edit-verification.js';
24
+ import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, verificationNudgeFor, THINK_ONLY_ESCALATION, AUTHORIZED_WORK_NUDGE, deliverableNudge, } from './edit-verification.js';
23
25
  import { effectiveToolJsonSchemas, coreToolJsonSchemas, isToolEnabled, toolsetForTool } from './toolsets.js';
26
+ import { deliverablesNamedIn, recordStepHandoff } from '../agents/step-handoff.js';
27
+ /**
28
+ * Tools that MUTATE the workspace. A refusal of one of these is the only
29
+ * refusal that leaves work undone — a declined `read_file` costs a step, a
30
+ * declined `write_file` means the deliverable was never produced.
31
+ */
32
+ const MUTATING_TOOL_NAMES = new Set([
33
+ 'write_file',
34
+ 'edit_file',
35
+ 'propose_change',
36
+ 'run_terminal',
37
+ 'run_cli',
38
+ ]);
39
+ /** The path a mutating call was aimed at, when it named one. */
40
+ function mutatedPathOf(args) {
41
+ const a = args;
42
+ const p = a?.path ?? a?.file_path ?? a?.file;
43
+ return typeof p === 'string' && p.trim() ? p.trim() : undefined;
44
+ }
45
+ /**
46
+ * The ask this turn is answering — the text a hand-off is keyed against.
47
+ * `authorizationRequest` is the loop's own record of the last user message;
48
+ * the thread scan is the fallback for callers that do not set it (tests, direct
49
+ * invocations). Injected blocks (`[Project context]`, the route feed) are
50
+ * skipped: they are the loop's own additions, not the user's words.
51
+ */
52
+ function currentAsk(opts) {
53
+ const recorded = opts.context?.authorizationRequest?.trim();
54
+ if (recorded)
55
+ return recorded.slice(0, 600);
56
+ const messages = opts.messages ?? [];
57
+ for (let i = messages.length - 1; i >= 0; i -= 1) {
58
+ const m = messages[i];
59
+ if (m.role !== 'user')
60
+ continue;
61
+ const text = (m.content ?? '').trim();
62
+ if (!text || text.startsWith('['))
63
+ continue;
64
+ return text.slice(0, 600);
65
+ }
66
+ return '';
67
+ }
24
68
  import { appendToolArtifact } from './artifact-append.js';
69
+ import { ROUTE_FEED_MARKER, isUnresolvedModel, routeFeedFingerprint, routeFeedText } from './loop-route-feed.js';
25
70
  import { logger } from '../utils/logger.js';
26
- import { toUserFacingGenerationError } from '../inference/tool-call-utils.js';
71
+ import { toUserFacingGenerationError, TEXTUAL_TOOL_CALL_HINT, isTextualFollowupsPayload, followupEntriesFromPayload, stripTrailingFollowupsHeader, } from '../inference/tool-call-utils.js';
27
72
  // ─── P3c — tool-fallback hints (switch tools when one fails) ────────────────
28
73
  // The ask: *"if one tool fails you switch to another and explore parallel
29
74
  // ways"* (round-2 row 25). Today the raw `Error: …` text is fed back and a
@@ -116,7 +161,8 @@ export function isThinkOnlyResponse(content) {
116
161
  }
117
162
  /**
118
163
  * Extract JSON fallback tool calls from model text:
119
- * `{"tool":"name","arguments":{...}}` blocks, possibly fenced or multiple.
164
+ * `{"tool":"name","arguments":{...}}` blocks, possibly fenced or multiple —
165
+ * plus the name-keyed `{"suggest_followups":[…]}` shape models actually write.
120
166
  * Uses brace-matching (string-aware) so nested argument objects parse
121
167
  * correctly. Returns the cleaned content (blocks stripped) + parsed calls.
122
168
  */
@@ -161,11 +207,59 @@ export function extractFallbackToolCalls(content) {
161
207
  // Unparseable block — dropped from the answer, no tool call.
162
208
  }
163
209
  }
210
+ // ── Pass 2: our tool's ARGUMENTS keyed by our own tool NAME ─────────────
211
+ // {"suggest_followups":[{"label":"…","prompt":"…"}, …]}
212
+ // This is the shape models ACTUALLY hand-write: across 116 stored assistant
213
+ // turns the canonical `{"tool":"…"}` form appeared ZERO times and this one
214
+ // 14. Pass 1 could not see it at all, so the block was delivered verbatim to
215
+ // the reader AND the suggestions were thrown away (no chips, no menu).
216
+ //
217
+ // The block is consumed in exactly two cases: it parses to OUR payload (the
218
+ // shared predicate — the same one the strip uses, so the two can never
219
+ // disagree), or it does not parse at all (truncated scaffolding is never
220
+ // prose). A value that parses cleanly to something demonstrably NOT the
221
+ // contract — `{"suggest_followups": "a note"}` — is left untouched, so a
222
+ // user's own JSON still survives.
223
+ const namedAt = /\(?\s*\{\s*"suggest_followups"\s*:/g;
224
+ let named;
225
+ while ((named = namedAt.exec(cleaned)) !== null) {
226
+ const end = findMatchingBrace(cleaned, named.index);
227
+ if (end === -1) {
228
+ // Cut off mid-call: everything from the marker on is scaffolding.
229
+ cleaned = cleaned.slice(0, named.index);
230
+ strippedAny = true;
231
+ break;
232
+ }
233
+ const block = cleaned.slice(named.index, end + 1);
234
+ let parsed = null;
235
+ let unparseable = false;
236
+ try {
237
+ parsed = JSON.parse(block);
238
+ }
239
+ catch {
240
+ unparseable = true;
241
+ }
242
+ if (!unparseable && !isTextualFollowupsPayload(parsed))
243
+ continue;
244
+ cleaned = cleaned.slice(0, named.index) + cleaned.slice(end + 1);
245
+ namedAt.lastIndex = named.index;
246
+ strippedAny = true;
247
+ const entries = followupEntriesFromPayload(parsed);
248
+ if (entries) {
249
+ calls.push({
250
+ id: `call_${calls.length + 1}`,
251
+ name: 'suggest_followups',
252
+ arguments: { followups: entries },
253
+ });
254
+ }
255
+ }
164
256
  if (!strippedAny)
165
257
  return { text: content, calls };
166
258
  // Remove empty fenced blocks left behind when a fenced JSON block's BODY was
167
- // the tool call (e.g. "```json\n\n```").
168
- const text = cleaned.replace(/\n?```[a-z]*\s*\n?\s*```\s*/gi, '\n').trim();
259
+ // the tool call (e.g. "```json\n\n```"), plus any caption/separator the model
260
+ // put in FRONT of the block. Without the second step the delivered answer
261
+ // ends on a dangling `---` — the visible half of the leak this recovers from.
262
+ const text = stripTrailingFollowupsHeader(cleaned.replace(/\n?```[a-z]*\s*\n?\s*```\s*/gi, '\n')).trim();
169
263
  return { text, calls };
170
264
  }
171
265
  /**
@@ -257,6 +351,16 @@ const MAX_PARALLEL_READS = 4;
257
351
  // half-done and resumable in-context. These defaults resume the SAME turn a
258
352
  // bounded number of times — the model keeps its thread, completed tool calls
259
353
  // are never re-run, and the loop can never spin forever.
354
+ /**
355
+ * Consecutive REASONING-ONLY steps the loop tolerates before it escalates, and
356
+ * then one more before it ends the turn.
357
+ *
358
+ * Small on purpose. A `<think>`-then-answer model needs a step or two, so the
359
+ * limit cannot be zero — but unbounded, this path is a spin: two live eval runs
360
+ * produced 31 reasoning-only steps, one tool call, and a 0% score on a task they
361
+ * were well able to do (see `THINK_ONLY_ESCALATION`).
362
+ */
363
+ export const MAX_THINK_CONTINUES = 3;
260
364
  /** Default continuations granted per turn when the option is omitted. */
261
365
  export const DEFAULT_MAX_CONTINUATIONS = 2;
262
366
  /** Default extra steps granted per continuation. */
@@ -296,6 +400,14 @@ async function runToolLoopInner(opts, progress) {
296
400
  // G13 — bounded "the request already authorized this" nudges (see
297
401
  // detectPermissionSeeking).
298
402
  let permissionNudges = 0;
403
+ // Stage 2 — the repetition nudge's own bound (see the block after the
404
+ // permission nudge). Separate from `permissionNudges` on purpose: that one
405
+ // asks the model to PROCEED, this one asks it to stop REPEATING, and the two
406
+ // fire on independent evidence.
407
+ let repeatNudges = 0;
408
+ // Bounded think-only continuation (see THINK_ONLY_ESCALATION). Counts the
409
+ // consecutive reasoning-only steps so the loop cannot spin on them.
410
+ let thinkContinues = 0;
299
411
  // G13b — bounded "the request asked for a file and none was written" nudges
300
412
  // (see wantsAuthoredArtifact).
301
413
  let deliverableNudges = 0;
@@ -342,6 +454,45 @@ async function runToolLoopInner(opts, progress) {
342
454
  // gate exactly as it was.
343
455
  const requestText = lastUserText(opts.messages);
344
456
  const authorization = requestAuthorizesWrites(requestText);
457
+ // ── INTENT ENVELOPE (the durable grant) ──────────────────────────────────
458
+ // `writesAuthorized` above answers "does the LAST message ask for files?" —
459
+ // once per turn, and forgotten at the turn boundary. That is why permission
460
+ // was re-asked for every operation and why an approval given minutes ago was
461
+ // worth nothing on the next turn. The envelope is the same verdict as a
462
+ // DURABLE, SCOPED object, keyed to the conversation (the dashboard console
463
+ // keeps one plan store per session and re-injects it every turn; the CLI keeps
464
+ // one per ChatCommand), so an approval outlives the turn that granted it.
465
+ //
466
+ // Grant order is deliberate: an APPROVED PLAN is the high-trust path (the user
467
+ // saw the plan and said yes), then a fresh directive request, then whatever is
468
+ // still live for this conversation.
469
+ const envelopeKey = isEnvelopeKey(context.planStore) ? context.planStore : undefined;
470
+ let envelope = getEnvelope(envelopeKey);
471
+ if (envelopeKey) {
472
+ const plan = context.planStore?.snapshot?.() ?? null;
473
+ if (plan && isAffirmativeReply(requestText)) {
474
+ envelope = envelopeFromPlan(plan.goal);
475
+ grantEnvelope(envelopeKey, envelope);
476
+ deps.onEvent?.(' ✅ Plan approved — running it under one grant; no per-step permission.');
477
+ }
478
+ else {
479
+ const fresh = envelopeFromRequest(requestText);
480
+ if (fresh) {
481
+ envelope = fresh;
482
+ grantEnvelope(envelopeKey, fresh);
483
+ }
484
+ }
485
+ }
486
+ // ── RUN TRACE (the loop's representation of its own behaviour) ────────────
487
+ // The envelope above answers "what may I do"; the trace answers "what have I
488
+ // been DOING" — the question nothing in the architecture could answer, which
489
+ // is why it could not notice its own loop and could not respond to a user
490
+ // asking about one. Keyed to the conversation like the envelope, with a
491
+ // per-turn fallback when there is no session handle so repetition WITHIN a
492
+ // single turn is still caught.
493
+ const traceKey = isTraceKey(context.planStore) ? context.planStore : undefined;
494
+ const runTrace = traceKey ? runTraceFor(traceKey) : new RunTrace();
495
+ progress.runTrace = runTrace;
345
496
  const ctx = {
346
497
  ...context,
347
498
  followups: context.followups || sink,
@@ -351,6 +502,11 @@ async function runToolLoopInner(opts, progress) {
351
502
  // ask for a commit? resolve to this CLI command?), and a single boolean
352
503
  // computed for a different question cannot answer any of them.
353
504
  authorizationRequest: requestText,
505
+ // The durable grant every gate consults FIRST (see learning/intent-envelope.ts).
506
+ envelope,
507
+ // The run's own behaviour, so a gate can refuse to RE-ASK a question the
508
+ // user already answered (see learning/run-trace.ts).
509
+ runTrace,
354
510
  };
355
511
  // Resolve the tool set — stable JSON schemas for every native step
356
512
  // (derived from the registry's zod schemas, single source of truth).
@@ -406,6 +562,30 @@ async function runToolLoopInner(opts, progress) {
406
562
  return added;
407
563
  };
408
564
  const thread = [...messages];
565
+ /** Serving pairs the loop has already told the model about, oldest first. */
566
+ const routeFeedState = { pairs: [] };
567
+ // ── Stage 2: answer a question ABOUT the run from the run ────────────────
568
+ // The live audit's worst moment was the user asking "why are you asking me
569
+ // this again and again?" and the turn replying with an edit plan — not through
570
+ // indifference, but because nothing held the fact that it had asked four
571
+ // times. Handing the trace in makes the honest answer possible; the framing
572
+ // line keeps it from being mistaken for a plan.
573
+ if (detectProcessComplaint(requestText)) {
574
+ deps.onEvent?.(' 🪞 The user asked about your own behaviour — answering from the run trace.');
575
+ traceEvent({
576
+ kind: 'gate',
577
+ gate: 'repeat',
578
+ summary: `the user asked about the agent's own behaviour — trace provided (${runTrace.countAsks()} ask(s), ${runTrace.repeatedAskCount()} repeated)`,
579
+ });
580
+ thread.push({
581
+ role: 'system',
582
+ content: `${runTrace.selfReport()}\n\n` +
583
+ 'The user is asking about THIS behaviour, not about the code. Answer them directly: ' +
584
+ 'state what you did (how many questions you asked, which one you repeated, and what they ' +
585
+ 'answered), acknowledge the repetition plainly if there was one, and say what you will do ' +
586
+ 'instead. Do not answer a question about your own process with a plan for the work.',
587
+ });
588
+ }
409
589
  const budgetChars = opts.threadBudgetChars ?? DEFAULT_THREAD_BUDGET_CHARS;
410
590
  // P3d — per-turn parallel suggester: after 2+ successful independent gather
411
591
  // steps, one advisory delegate suggestion fires (bounded, deterministic).
@@ -449,6 +629,12 @@ async function runToolLoopInner(opts, progress) {
449
629
  deps.onEvent?.(` ✂️ ${trimmedResult.trimmed} old tool result(s) trimmed to fit the ${Math.round(budgetChars / 1000)}K-char context budget.`);
450
630
  }
451
631
  }
632
+ // ── The route feed, re-synced before every call ────────────────────────
633
+ // Placed here rather than at thread construction because THIS is the last
634
+ // point before the call: a failover during step N must be visible at step
635
+ // N+1, and a single up-front line would have the model answering about a
636
+ // model that has since stopped serving it.
637
+ syncRouteFeed(thread, opts.servedRoute, routeFeedState);
452
638
  let response;
453
639
  try {
454
640
  response = await deps.callModel(thread, schemas, opts.onToken, opts.signal);
@@ -542,7 +728,7 @@ async function runToolLoopInner(opts, progress) {
542
728
  // console, gateway) recovers the calls, and strip the block from the
543
729
  // visible answer either way: a raw `{"tool":…}` block must never be the
544
730
  // answer. Idempotent with the JSON transport (which already strips it).
545
- if (response.toolCalls.length === 0 && response.content.includes('"tool"')) {
731
+ if (response.toolCalls.length === 0 && TEXTUAL_TOOL_CALL_HINT.test(response.content)) {
546
732
  const extracted = extractFallbackToolCalls(response.content);
547
733
  if (extracted.text !== response.content) {
548
734
  response.content = extracted.text;
@@ -574,12 +760,63 @@ async function runToolLoopInner(opts, progress) {
574
760
  if (deps.isThinkOnly ? deps.isThinkOnly(response.content) : isThinkOnlyResponse(response.content)) {
575
761
  // Feed an empty assistant step so the model continues in-context.
576
762
  thread.push({ role: 'assistant', content: response.content });
577
- deps.onEvent?.(' 🧠 model reasoning… (continuing)');
763
+ thinkContinues += 1;
764
+ // AN EMPTY RESPONSE IS A FAILURE, NOT REASONING. `isThinkOnlyResponse('')`
765
+ // returns true, so a provider returning nothing (the loop arm falls over
766
+ // to `local/gpt-oss:120b-cloud`, which reliably returns length 0) was
767
+ // reported as "model reasoning… (continuing)" — the system's own evidence
768
+ // saying the agent was thinking when it was being handed nothing. That
769
+ // is the exact class of defect this work exists to remove, so the two
770
+ // cases are named apart even though they are bounded together.
771
+ const emptyResponse = response.content.trim().length === 0;
772
+ // BOUNDED (see THINK_ONLY_ESCALATION). Continuing on reasoning is right
773
+ // for a `<think>`-then-answer model and catastrophic without a limit:
774
+ // two live eval runs produced 31 reasoning-only steps, one tool call and
775
+ // a 0% score, printing "model reasoning… (continuing)" the whole way.
776
+ if (thinkContinues <= MAX_THINK_CONTINUES) {
777
+ deps.onEvent?.(emptyResponse
778
+ ? ` ⚠️ the provider returned an EMPTY response (${thinkContinues}/${MAX_THINK_CONTINUES}) — retrying the step.`
779
+ : ` 🧠 model reasoning… (continuing ${thinkContinues}/${MAX_THINK_CONTINUES})`);
780
+ traceEvent({
781
+ kind: 'decision',
782
+ summary: emptyResponse
783
+ ? 'the provider returned an empty response — no answer text and no tool call; the step was retried'
784
+ : 'the model replied with its own reasoning instead of an answer — the step continued',
785
+ });
786
+ continue;
787
+ }
788
+ if (thinkContinues === MAX_THINK_CONTINUES + 1) {
789
+ // One escalation: it has no answer yet, so tell it to act or answer.
790
+ stepLimit += 1;
791
+ deps.onEvent?.(emptyResponse
792
+ ? ' ⚠️ The provider keeps returning empty responses — asking for one real step.'
793
+ : ' 🧠 Reasoning only, repeatedly — telling the model to act or answer now.');
794
+ traceEvent({
795
+ kind: 'gate',
796
+ gate: 'repeat',
797
+ summary: emptyResponse
798
+ ? `the provider returned an empty response ${thinkContinues} times — one bounded escalation for a real step`
799
+ : `the model produced reasoning-only output ${thinkContinues} times with no tool call and no answer — ` +
800
+ 'one bounded escalation to act or answer',
801
+ });
802
+ thread.push({ role: 'user', content: THINK_ONLY_ESCALATION });
803
+ continue;
804
+ }
805
+ // Still spinning after the escalation: END the turn rather than burn the
806
+ // rest of the budget. `bounded` is set so every surface reads this as
807
+ // "stopped before finishing", never as a completed answer.
808
+ bounded = true;
809
+ deps.onEvent?.(emptyResponse
810
+ ? ' ⚠️ Empty responses kept coming — ending the turn instead of spinning. This is a PROVIDER failure, not the agent thinking.'
811
+ : ' 🧠 Reasoning-only output kept repeating — ending the turn instead of spinning.');
578
812
  traceEvent({
579
- kind: 'decision',
580
- summary: 'the model replied with its own reasoning instead of an answer — the step continued',
813
+ kind: 'gate',
814
+ gate: 'repeat',
815
+ summary: emptyResponse
816
+ ? `the provider returned ${thinkContinues} consecutive empty responses — the turn ended; this is a transport failure, not agent stuckness`
817
+ : `reasoning-only output repeated ${thinkContinues} times after an escalation — the turn ended to avoid an unbounded spin`,
581
818
  });
582
- continue;
819
+ break;
583
820
  }
584
821
  // ── Dangling-promise nudge (bounded, once) ──────────────────────────
585
822
  // The model closed the turn announcing what it is ABOUT to do ("I will
@@ -654,6 +891,39 @@ async function runToolLoopInner(opts, progress) {
654
891
  thread.push({ role: 'user', content: AUTHORIZED_WORK_NUDGE });
655
892
  continue;
656
893
  }
894
+ // ── Stage 2 — REPETITION nudge (bounded, once) ───────────────────────
895
+ // The permission nudge above fires only when the request AUTHORIZED the
896
+ // work, and it asks the model to PROCEED. A repeated question is a defect
897
+ // on its own terms: the user already answered it, whatever their latest
898
+ // message authorized — which is precisely the case a live turn fell
899
+ // through, because the complaint about repeated questions was itself the
900
+ // message that failed to authorize. This nudge therefore does NOT depend
901
+ // on authorization, and deliberately does NOT say "proceed" (unsafe when
902
+ // the answer was no) — it says the answer is in hand, so act on it or say
903
+ // what blocks you. The evidence is the RUN TRACE, not the current text.
904
+ if (repeatNudges < 1 &&
905
+ schemas.length > 0 &&
906
+ detectPermissionSeeking(response.content.length >= lastContent.length ? response.content : lastContent) &&
907
+ runTrace.priorAskMatches(response.content.length >= lastContent.length ? response.content : lastContent).length > 0) {
908
+ const closing = response.content.length >= lastContent.length ? response.content : lastContent;
909
+ repeatNudges += 1;
910
+ stepLimit += 1;
911
+ deps.onEvent?.(' 🔁 The turn asked a question the user has already answered — telling the model to act on the answer.');
912
+ traceEvent({
913
+ kind: 'gate',
914
+ gate: 'repeat',
915
+ summary: 'the turn repeated a question already asked and answered in this conversation — one bounded nudge to use the answer',
916
+ });
917
+ if (progress.successfulToolCalls.length === 0) {
918
+ lastContent = '';
919
+ }
920
+ else {
921
+ lastContent = stripTrailingPermissionSeek(closing);
922
+ }
923
+ thread.push({ role: 'assistant', content: response.content });
924
+ thread.push({ role: 'user', content: repeatNudge(closing, runTrace.priorAnswer(closing)) });
925
+ continue;
926
+ }
657
927
  // S1 (both exits): the MOST SUBSTANTIVE content seen wins here too —
658
928
  // a short closing step ("Sent it to her! ✅") with no tool calls must
659
929
  // not clobber the deliverable (poem/essay) the model composed in an
@@ -700,7 +970,14 @@ async function runToolLoopInner(opts, progress) {
700
970
  summary: 'the turn mutated the workspace and nothing observed the result — one bounded nudge to verify',
701
971
  });
702
972
  thread.push({ role: 'assistant', content: response.content });
703
- thread.push({ role: 'user', content: VERIFICATION_NUDGE });
973
+ // Stage 3 — name THIS project's strongest check instead of describing a
974
+ // preference order. A live turn reached for `node -c` because the nudge
975
+ // left the choice to the model; `verificationNudgeFor` reads the
976
+ // workspace and asks for the real command, so there is nothing to guess.
977
+ thread.push({
978
+ role: 'user',
979
+ content: verificationNudgeFor(ctx.cwd ?? process.cwd(), progress.mutatedPaths),
980
+ });
704
981
  continue;
705
982
  }
706
983
  return {
@@ -902,13 +1179,59 @@ async function runToolLoopInner(opts, progress) {
902
1179
  if (call.name === 'edit_file' || call.name === 'write_file') {
903
1180
  const a = call.arguments;
904
1181
  const p = a?.path ?? a?.file_path ?? a?.file;
905
- if (typeof p === 'string' && p)
1182
+ if (typeof p === 'string' && p) {
906
1183
  progress.mutatedPaths.push(p);
1184
+ // Stage 2 — the run's own record of what it CHANGED, so a self-report
1185
+ // can answer "what have you been doing" from data rather than from a
1186
+ // re-read of the transcript.
1187
+ runTrace.recordMutation(call.name, p);
1188
+ }
907
1189
  }
908
1190
  else if (isVerificationTool(call.name)) {
909
1191
  progress.verificationEvidence.push({ tool: call.name, args: call.arguments, result: rawResult });
910
1192
  }
911
1193
  }
1194
+ else if (refusal !== null || rawResult.startsWith('Error:')) {
1195
+ // Stage 2 — the other half of noticing a loop: the SAME refusal, twice.
1196
+ // The run records it with its reason, so a later step (or a self-report)
1197
+ // can see "blocked twice, identically" instead of rediscovering it — and
1198
+ // so the repetition gate has a refusal signal as well as an ask signal.
1199
+ runTrace.recordRefusal(call.name, refusal?.gate ?? 'error', rawResult);
1200
+ // ── DURABLE HAND-OFF ────────────────────────────────────────────────
1201
+ // A refused MUTATION is the one failure that must outlive the turn.
1202
+ // The live NVDA-addon failure is exactly this: `write_file` was refused
1203
+ // on every attempt because the target was outside the workspace, and
1204
+ // each of the 18 following attempts rediscovered it from zero — same
1205
+ // plan, same refusal, same empty package. Recording it here means the
1206
+ // next candidate, the next turn and the next RUN are told which path
1207
+ // could not be written and WHY, so they change approach instead of
1208
+ // repeating the identical call (see step-handoff.ts).
1209
+ //
1210
+ // Best-effort: a hand-off write must never break the loop.
1211
+ if (refusal !== null && MUTATING_TOOL_NAMES.has(call.name)) {
1212
+ try {
1213
+ const ask = currentAsk(opts);
1214
+ const aimedAt = mutatedPathOf(call.arguments);
1215
+ // Key on the deliverable the ASK names when there is one, so the
1216
+ // same work asked for in different words resumes the same hand-off.
1217
+ const named = deliverablesNamedIn(ask);
1218
+ recordStepHandoff({
1219
+ projectPath: opts.context?.cwd || process.cwd(),
1220
+ goal: ask,
1221
+ stepDescription: aimedAt
1222
+ ? `${call.name} → ${aimedAt} (refused: ${refusal.gate ?? 'gate'})`
1223
+ : `${call.name} (refused: ${refusal.gate ?? 'gate'})`,
1224
+ declared: named.length > 0 ? named : aimedAt ? [aimedAt] : [],
1225
+ route: 'loop',
1226
+ kind: 'refused',
1227
+ reason: `${refusal.summary}${aimedAt ? ` — path: ${aimedAt}` : ''}`,
1228
+ });
1229
+ }
1230
+ catch {
1231
+ // Best-effort — a hand-off write must never break the loop.
1232
+ }
1233
+ }
1234
+ }
912
1235
  let resultText = executed[i];
913
1236
  // P3c — on error/denial, append the deterministic fallback hint for
914
1237
  // this tool (advisory — the model still decides; never on success).
@@ -1019,7 +1342,14 @@ async function runToolLoopInner(opts, progress) {
1019
1342
  summary: 'the turn mutated the workspace and nothing observed the result — one bounded nudge to verify',
1020
1343
  });
1021
1344
  thread.push({ role: 'assistant', content: response.content });
1022
- thread.push({ role: 'user', content: VERIFICATION_NUDGE });
1345
+ // Stage 3 — name THIS project's strongest check instead of describing a
1346
+ // preference order. A live turn reached for `node -c` because the nudge
1347
+ // left the choice to the model; `verificationNudgeFor` reads the
1348
+ // workspace and asks for the real command, so there is nothing to guess.
1349
+ thread.push({
1350
+ role: 'user',
1351
+ content: verificationNudgeFor(ctx.cwd ?? process.cwd(), progress.mutatedPaths),
1352
+ });
1023
1353
  continue;
1024
1354
  }
1025
1355
  // G13b — DELIVERABLE GATE (concluding path). AFTER the verification gate
@@ -1199,6 +1529,69 @@ export function detectUnfulfilledIntentPromise(content) {
1199
1529
  * honest-answer flags so every surface (CLI, dashboard, gateway, trace) can
1200
1530
  * distinguish "generated a reply" from "actually performed the action".
1201
1531
  */
1532
+ /**
1533
+ * Keep ONE route frame in the thread, current with what is serving the turn.
1534
+ *
1535
+ * Insert/replace is deliberate: appending a frame per step would pile up stale
1536
+ * route claims the model can quote back ("I am gemini-2.5-flash" — from step 1,
1537
+ * after a failover), and rewriting unconditionally would churn the thread (and
1538
+ * the thread budget) on every step for no change in content.
1539
+ *
1540
+ * A caller with no route supplies nothing; a caller whose route resolves to the
1541
+ * same fact as last step gets no write.
1542
+ */
1543
+ function syncRouteFeed(thread, read, observed) {
1544
+ if (!read)
1545
+ return;
1546
+ let route = null;
1547
+ try {
1548
+ route = read();
1549
+ }
1550
+ catch {
1551
+ // A route read must never break the turn; an unknown route is a fact too.
1552
+ return;
1553
+ }
1554
+ if (!route)
1555
+ return;
1556
+ // The loop OWNS the failover history, because the loop is the only thing that
1557
+ // sees the sequence of routes it told the model about. Requiring each caller
1558
+ // to track it means every future caller can silently forget, and a model told
1559
+ // only the current pair describes the whole turn as having been run by it —
1560
+ // the same fabricated answer, one layer down.
1561
+ const currentPair = isUnresolvedModel(route.model) ? '' : `${route.providerType}/${route.model}`;
1562
+ const previous = [...(route.previous ?? [])];
1563
+ for (const pair of observed.pairs) {
1564
+ if (pair !== currentPair && !previous.includes(pair))
1565
+ previous.push(pair);
1566
+ }
1567
+ const rendered = previous.length > 0 ? { ...route, previous } : route;
1568
+ const fingerprint = routeFeedFingerprint(rendered);
1569
+ const index = thread.findIndex((m) => m.role === 'system' && typeof m.content === 'string' && m.content.startsWith(ROUTE_FEED_MARKER));
1570
+ if (currentPair && !observed.pairs.includes(currentPair))
1571
+ observed.pairs.push(currentPair);
1572
+ if (index === -1) {
1573
+ const frame = { role: 'system', content: routeFeedText(rendered) };
1574
+ // Immediately after the leading system prompt when there is one — the route
1575
+ // belongs with the run's own framing, not buried under tool output.
1576
+ const insertAt = thread.length > 0 && thread[0].role === 'system' ? 1 : 0;
1577
+ thread.splice(insertAt, 0, frame);
1578
+ routeFeedFingerprints.set(frame, fingerprint);
1579
+ return;
1580
+ }
1581
+ const existing = thread[index];
1582
+ if (routeFeedFingerprints.get(existing) === fingerprint)
1583
+ return;
1584
+ const updated = { role: 'system', content: routeFeedText(rendered) };
1585
+ thread[index] = updated;
1586
+ routeFeedFingerprints.delete(existing);
1587
+ routeFeedFingerprints.set(updated, fingerprint);
1588
+ }
1589
+ /**
1590
+ * Fingerprint per frame object, so a thread that was trimmed and re-built does
1591
+ * not re-write identical content. A WeakMap: a frame the loop drops is garbage
1592
+ * the moment nothing references it.
1593
+ */
1594
+ const routeFeedFingerprints = new WeakMap();
1202
1595
  export async function runToolLoop(opts) {
1203
1596
  const progress = {
1204
1597
  successfulToolCalls: [],
@@ -1210,6 +1603,8 @@ export async function runToolLoop(opts) {
1210
1603
  if (!result.cancelled && !result.generationFailed) {
1211
1604
  result.successfulToolCalls = [...progress.successfulToolCalls];
1212
1605
  result.deliveryConfirmed = progress.deliveryConfirmed;
1606
+ // Stage 2 — the turn's own behaviour, as counts (see learning/run-trace.ts).
1607
+ result.runTrace = progress.runTrace?.snapshot();
1213
1608
  // Judge the "I have sent it" claim against what actually DELIVERED, not
1214
1609
  // against the attempted tool list — a failed gateway_send must not silence
1215
1610
  // the honesty correction.