agent-nuvira 3.3.9 → 3.3.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (153) hide show
  1. package/dist/agents/orchestrator.d.ts.map +1 -1
  2. package/dist/agents/orchestrator.js +6 -0
  3. package/dist/agents/orchestrator.js.map +1 -1
  4. package/dist/cli/chat.d.ts.map +1 -1
  5. package/dist/cli/chat.js +27 -17
  6. package/dist/cli/chat.js.map +1 -1
  7. package/dist/cli/config.d.ts +28 -0
  8. package/dist/cli/config.d.ts.map +1 -1
  9. package/dist/cli/config.js +155 -0
  10. package/dist/cli/config.js.map +1 -1
  11. package/dist/cli/eval.d.ts +8 -0
  12. package/dist/cli/eval.d.ts.map +1 -1
  13. package/dist/cli/eval.js +64 -0
  14. package/dist/cli/eval.js.map +1 -1
  15. package/dist/cli/loop-executor.d.ts +13 -0
  16. package/dist/cli/loop-executor.d.ts.map +1 -1
  17. package/dist/cli/loop-executor.js +22 -18
  18. package/dist/cli/loop-executor.js.map +1 -1
  19. package/dist/config/capability-mode.d.ts +122 -0
  20. package/dist/config/capability-mode.d.ts.map +1 -0
  21. package/dist/config/capability-mode.js +132 -0
  22. package/dist/config/capability-mode.js.map +1 -0
  23. package/dist/config/limits.d.ts +47 -0
  24. package/dist/config/limits.d.ts.map +1 -0
  25. package/dist/config/limits.js +64 -0
  26. package/dist/config/limits.js.map +1 -0
  27. package/dist/config/process-env.d.ts.map +1 -1
  28. package/dist/config/process-env.js +44 -0
  29. package/dist/config/process-env.js.map +1 -1
  30. package/dist/config/types.d.ts +77 -0
  31. package/dist/config/types.d.ts.map +1 -1
  32. package/dist/config/work-digest.d.ts +46 -0
  33. package/dist/config/work-digest.d.ts.map +1 -0
  34. package/dist/config/work-digest.js +80 -0
  35. package/dist/config/work-digest.js.map +1 -0
  36. package/dist/gateway/inbound-media.d.ts +18 -0
  37. package/dist/gateway/inbound-media.d.ts.map +1 -1
  38. package/dist/gateway/inbound-media.js +96 -2
  39. package/dist/gateway/inbound-media.js.map +1 -1
  40. package/dist/inference/anthropic-adapter.d.ts +6 -0
  41. package/dist/inference/anthropic-adapter.d.ts.map +1 -1
  42. package/dist/inference/anthropic-adapter.js +79 -8
  43. package/dist/inference/anthropic-adapter.js.map +1 -1
  44. package/dist/inference/gemini-adapter.d.ts +6 -0
  45. package/dist/inference/gemini-adapter.d.ts.map +1 -1
  46. package/dist/inference/gemini-adapter.js +73 -8
  47. package/dist/inference/gemini-adapter.js.map +1 -1
  48. package/dist/inference/groq-adapter.d.ts +3 -0
  49. package/dist/inference/groq-adapter.d.ts.map +1 -1
  50. package/dist/inference/groq-adapter.js +83 -40
  51. package/dist/inference/groq-adapter.js.map +1 -1
  52. package/dist/inference/interface.d.ts +7 -0
  53. package/dist/inference/interface.d.ts.map +1 -1
  54. package/dist/inference/model-probe.d.ts +17 -0
  55. package/dist/inference/model-probe.d.ts.map +1 -1
  56. package/dist/inference/model-probe.js +58 -1
  57. package/dist/inference/model-probe.js.map +1 -1
  58. package/dist/inference/model-validator.d.ts +16 -1
  59. package/dist/inference/model-validator.d.ts.map +1 -1
  60. package/dist/inference/model-validator.js +83 -2
  61. package/dist/inference/model-validator.js.map +1 -1
  62. package/dist/inference/nim-adapter.js +4 -4
  63. package/dist/inference/nim-adapter.js.map +1 -1
  64. package/dist/inference/openai-compat-adapter.d.ts +9 -0
  65. package/dist/inference/openai-compat-adapter.d.ts.map +1 -1
  66. package/dist/inference/openai-compat-adapter.js +101 -51
  67. package/dist/inference/openai-compat-adapter.js.map +1 -1
  68. package/dist/inference/openrouter-adapter.d.ts +3 -0
  69. package/dist/inference/openrouter-adapter.d.ts.map +1 -1
  70. package/dist/inference/openrouter-adapter.js +134 -57
  71. package/dist/inference/openrouter-adapter.js.map +1 -1
  72. package/dist/inference/reasoning-effort.d.ts +130 -0
  73. package/dist/inference/reasoning-effort.d.ts.map +1 -0
  74. package/dist/inference/reasoning-effort.js +237 -0
  75. package/dist/inference/reasoning-effort.js.map +1 -0
  76. package/dist/inference/route-resolver.d.ts +7 -0
  77. package/dist/inference/route-resolver.d.ts.map +1 -1
  78. package/dist/inference/route-resolver.js +2 -1
  79. package/dist/inference/route-resolver.js.map +1 -1
  80. package/dist/inference/sse.d.ts +6 -0
  81. package/dist/inference/sse.d.ts.map +1 -1
  82. package/dist/inference/sse.js +1 -0
  83. package/dist/inference/sse.js.map +1 -1
  84. package/dist/inference/tools.d.ts +13 -0
  85. package/dist/inference/tools.d.ts.map +1 -1
  86. package/dist/inference/tools.js +20 -2
  87. package/dist/inference/tools.js.map +1 -1
  88. package/dist/learning/autonomy-policy.d.ts +8 -0
  89. package/dist/learning/autonomy-policy.d.ts.map +1 -1
  90. package/dist/learning/autonomy-policy.js +34 -0
  91. package/dist/learning/autonomy-policy.js.map +1 -1
  92. package/dist/learning/capability-parity.d.ts +108 -0
  93. package/dist/learning/capability-parity.d.ts.map +1 -0
  94. package/dist/learning/capability-parity.js +154 -0
  95. package/dist/learning/capability-parity.js.map +1 -0
  96. package/dist/learning/cost-tracker.d.ts +28 -4
  97. package/dist/learning/cost-tracker.d.ts.map +1 -1
  98. package/dist/learning/cost-tracker.js +58 -8
  99. package/dist/learning/cost-tracker.js.map +1 -1
  100. package/dist/learning/eval-framework.d.ts +17 -0
  101. package/dist/learning/eval-framework.d.ts.map +1 -1
  102. package/dist/learning/eval-framework.js +63 -2
  103. package/dist/learning/eval-framework.js.map +1 -1
  104. package/dist/learning/model-capability.d.ts +94 -0
  105. package/dist/learning/model-capability.d.ts.map +1 -0
  106. package/dist/learning/model-capability.js +172 -0
  107. package/dist/learning/model-capability.js.map +1 -0
  108. package/dist/learning/model-registry.d.ts +33 -0
  109. package/dist/learning/model-registry.d.ts.map +1 -1
  110. package/dist/learning/model-registry.js +79 -0
  111. package/dist/learning/model-registry.js.map +1 -1
  112. package/dist/learning/reasoning-trace.d.ts +1 -1
  113. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  114. package/dist/learning/reasoning-trace.js.map +1 -1
  115. package/dist/learning/resolve-options.d.ts.map +1 -1
  116. package/dist/learning/resolve-options.js +16 -3
  117. package/dist/learning/resolve-options.js.map +1 -1
  118. package/dist/learning/run-trace.d.ts +68 -0
  119. package/dist/learning/run-trace.d.ts.map +1 -1
  120. package/dist/learning/run-trace.js +101 -0
  121. package/dist/learning/run-trace.js.map +1 -1
  122. package/dist/tools/loop-skill-hint.d.ts +27 -0
  123. package/dist/tools/loop-skill-hint.d.ts.map +1 -1
  124. package/dist/tools/loop-skill-hint.js +189 -5
  125. package/dist/tools/loop-skill-hint.js.map +1 -1
  126. package/dist/tools/read-extract.d.ts +9 -3
  127. package/dist/tools/read-extract.d.ts.map +1 -1
  128. package/dist/tools/read-extract.js +70 -25
  129. package/dist/tools/read-extract.js.map +1 -1
  130. package/dist/tools/tool-loop.d.ts +120 -0
  131. package/dist/tools/tool-loop.d.ts.map +1 -1
  132. package/dist/tools/tool-loop.js +447 -5
  133. package/dist/tools/tool-loop.js.map +1 -1
  134. package/dist/web-dashboard/attachment-extract.d.ts +8 -1
  135. package/dist/web-dashboard/attachment-extract.d.ts.map +1 -1
  136. package/dist/web-dashboard/attachment-extract.js +14 -3
  137. package/dist/web-dashboard/attachment-extract.js.map +1 -1
  138. package/dist/web-dashboard/server.d.ts.map +1 -1
  139. package/dist/web-dashboard/server.js +160 -21
  140. package/dist/web-dashboard/server.js.map +1 -1
  141. package/dist/web-dashboard/src/types.d.ts +26 -0
  142. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  143. package/dist/web-dashboard/workspace-guard.d.ts.map +1 -1
  144. package/dist/web-dashboard/workspace-guard.js +44 -4
  145. package/dist/web-dashboard/workspace-guard.js.map +1 -1
  146. package/package.json +1 -1
  147. package/src/web-dashboard/public/assets/index-BdKf5Xw2.js +207 -0
  148. package/src/web-dashboard/public/assets/index-BdKf5Xw2.js.map +1 -0
  149. package/src/web-dashboard/public/assets/index-C9eBskc9.css +1 -0
  150. package/src/web-dashboard/public/index.html +2 -2
  151. package/src/web-dashboard/public/assets/index-Beportyl.js +0 -207
  152. package/src/web-dashboard/public/assets/index-Beportyl.js.map +0 -1
  153. package/src/web-dashboard/public/assets/index-gtQyg9nm.css +0 -1
@@ -33,11 +33,18 @@ import { stepDigest } from '../learning/step-checkpoint.js';
33
33
  // process declared no fault (`NUVIRA_INJECT_FAULT`), so an ordinary turn is
34
34
  // unaffected.
35
35
  import { faultAt } from '../runtime/fault-injection.js';
36
- import { detectPermissionSeeking, isAffirmativeReply, replyAsksTheReader, requestAuthorizesWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
36
+ import { detectPermissionSeeking, isAffirmativeReply, replyAsksTheReader, requestAuthorizesWrites, requestForbidsWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
37
37
  import { envelopeFromPlan, envelopeFromRequest, getEnvelope, grantEnvelope, isEnvelopeKey, } from '../learning/intent-envelope.js';
38
- import { detectProcessComplaint, isTraceKey, repeatNudge, runTraceFor, RunTrace, } from '../learning/run-trace.js';
38
+ import { detectProcessComplaint, isTraceKey, noProgressNudge, repeatNudge, repeatedFailureNudge, runTraceFor, RunTrace, } from '../learning/run-trace.js';
39
39
  import { wantsAuthoredArtifact } from '../learning/deliverable-class.js';
40
40
  import { normalizeFollowups } from './followup-utils.js';
41
+ // The capability mode (balanced | max) — `max` widens the loop's own reasoning
42
+ // budget as well as routing, so "cost is not a concern" means the agent may
43
+ // keep working through a long build instead of stopping at the default bound.
44
+ import { isMaxCapability } from '../config/capability-mode.js';
45
+ // The digest's scope (all | max | off) — a user-facing control over whether the
46
+ // compaction digest is injected in every mode, only under `max`, or never.
47
+ import { isWorkDigestEnabled } from '../config/work-digest.js';
41
48
  import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, verificationNudgeFor, THINK_ONLY_ESCALATION, AUTHORIZED_WORK_NUDGE, deliverableNudge, } from './edit-verification.js';
42
49
  import { effectiveToolJsonSchemas, coreToolJsonSchemas, isToolEnabled, toolsetForTool } from './toolsets.js';
43
50
  import { deliverablesNamedIn, recordStepHandoff } from '../agents/step-handoff.js';
@@ -60,6 +67,22 @@ function mutatedPathOf(args) {
60
67
  const p = a?.path ?? a?.file_path ?? a?.file;
61
68
  return typeof p === 'string' && p.trim() ? p.trim() : undefined;
62
69
  }
70
+ /**
71
+ * The ACTION signature of a failed call — what it tried to do, so the identical
72
+ * retry is recognizable. Prefers the shell `command`, then the file `path`,
73
+ * then the tool name (a call that named neither still repeats as the same tool).
74
+ * Used by the self-diagnosis gate (see RunTrace.repeatedFailure).
75
+ */
76
+ function failureActionOf(call) {
77
+ const a = call.arguments;
78
+ const command = typeof a?.command === 'string' && a.command.trim() ? a.command.trim() : undefined;
79
+ if (command)
80
+ return command;
81
+ const path = mutatedPathOf(call.arguments);
82
+ if (path)
83
+ return path;
84
+ return call.name;
85
+ }
63
86
  /**
64
87
  * Append what a tool call actually did to the turn's provenance ledger.
65
88
  *
@@ -396,10 +419,38 @@ const MAX_PARALLEL_READS = 4;
396
419
  * were well able to do (see `THINK_ONLY_ESCALATION`).
397
420
  */
398
421
  export const MAX_THINK_CONTINUES = 3;
422
+ /**
423
+ * Consecutive steps that RAN tools and had NO success before the loop asks the
424
+ * model to diagnose the stall. This generalises {@link RunTrace.repeatedFailure}
425
+ * (the identical action retried): a weak model also loops by substituting a
426
+ * different command each step, every one failing the same underlying way — none
427
+ * of them repeats exactly, so the same-action gate never fires.
428
+ *
429
+ * The threshold is deliberately ABOVE the handful of distinct checks ordinary
430
+ * iteration tries (`npm test` → `npm run build` → `npm run lint`), so a short run
431
+ * of unrelated failures is left alone; only a sustained all-fail streak is a stall.
432
+ */
433
+ export const NO_PROGRESS_STALL_STEPS = 4;
434
+ /**
435
+ * How substantial a turn must be before the SELF-REVIEW gate may fire (see
436
+ * `selfReviewNudge`). Short, single-edit turns already end with an obvious
437
+ * result the model just looked at; a long turn is where scope drift hides, and
438
+ * where an extra pass earns its latency.
439
+ */
440
+ export const SELF_REVIEW_MIN_STEPS = 6;
399
441
  /** Default continuations granted per turn when the option is omitted. */
400
442
  export const DEFAULT_MAX_CONTINUATIONS = 2;
401
443
  /** Default extra steps granted per continuation. */
402
444
  export const DEFAULT_CONTINUATION_STEPS = 8;
445
+ /**
446
+ * `max` capability mode — the loop's longest bounded reasoning budget. The
447
+ * defaults above are sized to keep an ordinary turn responsive; under `max` the
448
+ * user has said cost is not a concern, so the same session is allowed to run a
449
+ * long build / many-file edit to completion instead of stopping at the default
450
+ * bound. Still finite: the hard cap is `maxSteps + 4 * 16` extra steps.
451
+ */
452
+ export const MAX_CAPABILITY_MAX_CONTINUATIONS = 4;
453
+ export const MAX_CAPABILITY_CONTINUATION_STEPS = 16;
403
454
  /** Pause before re-attempting a failed step (lets a transient outage clear). */
404
455
  export const CONTINUATION_DELAY_MS = 1_500;
405
456
  /**
@@ -425,8 +476,12 @@ function sleep(ms) {
425
476
  async function runToolLoopInner(opts, progress) {
426
477
  const { messages, tools: toolNames, maxSteps = 16, context, deps } = opts;
427
478
  // Bounded auto-continuation state (see ToolLoopOptions.maxContinuations).
428
- const maxContinuations = Math.max(0, opts.maxContinuations ?? DEFAULT_MAX_CONTINUATIONS);
429
- const continuationSteps = Math.max(1, opts.continuationSteps ?? DEFAULT_CONTINUATION_STEPS);
479
+ // `max` capability mode raises the DEFAULT budget (an explicit option still
480
+ // wins, so callers that pin a bound keep it). Read once per turn from the
481
+ // config, exactly like routing, so a mode change applies to the next turn.
482
+ const maxCapability = isMaxCapability(context.configManager);
483
+ const maxContinuations = Math.max(0, opts.maxContinuations ?? (maxCapability ? MAX_CAPABILITY_MAX_CONTINUATIONS : DEFAULT_MAX_CONTINUATIONS));
484
+ const continuationSteps = Math.max(1, opts.continuationSteps ?? (maxCapability ? MAX_CAPABILITY_CONTINUATION_STEPS : DEFAULT_CONTINUATION_STEPS));
430
485
  let continuations = 0;
431
486
  // Bounded dangling-promise nudges spent this turn (see INTENT_PROMISE_RE).
432
487
  let intentNudges = 0;
@@ -440,12 +495,26 @@ async function runToolLoopInner(opts, progress) {
440
495
  // asks the model to PROCEED, this one asks it to stop REPEATING, and the two
441
496
  // fire on independent evidence.
442
497
  let repeatNudges = 0;
498
+ // The loop's SELF-DIAGNOSIS nudge: bounded once per turn, fired when the SAME
499
+ // action has failed more than once (see RunTrace.repeatedFailure). This is the
500
+ // capability that turns "retry the identical failing command 10 times" into
501
+ // "state the root cause and change the approach" — the live macOS-build turn
502
+ // re-issued `npx tauri build` again and again against the same missing Cargo.
503
+ let diagnosisNudges = 0;
504
+ // Stage 2 — consecutive tool-running steps with NO success (the generalized
505
+ // stall signal, see NO_PROGRESS_STALL_STEPS). Reset by any successful tool.
506
+ let failedToolSteps = 0;
507
+ // Stage 4 — bounded SELF-REVIEW nudge (see requireSelfReview / selfReviewNudge).
508
+ let selfReviewNudges = 0;
443
509
  // Bounded think-only continuation (see THINK_ONLY_ESCALATION). Counts the
444
510
  // consecutive reasoning-only steps so the loop cannot spin on them.
445
511
  let thinkContinues = 0;
446
512
  // G13b — bounded "the request asked for a file and none was written" nudges
447
513
  // (see wantsAuthoredArtifact).
448
514
  let deliverableNudges = 0;
515
+ // ZERO-ACTION — bounded "the request directed work and NOTHING was done"
516
+ // nudges (see requireAction / zeroActionGateApplies).
517
+ let actionNudges = 0;
449
518
  /**
450
519
  * G18 — the sink, wrapped so an observability failure can never become a
451
520
  * turn failure (a recorder that throws is a bug in the instrument, not in
@@ -646,6 +715,27 @@ async function runToolLoopInner(opts, progress) {
646
715
  // with an empty/bounded response (the "where is the essay?" bug).
647
716
  let lastContent = '';
648
717
  let bounded = false;
718
+ /**
719
+ * Should the SELF-REVIEW gate fire now? Returns the correction to inject, or
720
+ * null. Extracted so the two end-of-turn exits ask the question identically —
721
+ * nothing here is a function of which branch is ending.
722
+ */
723
+ const selfReviewCorrection = () => {
724
+ if (selfReviewNudges >= 1)
725
+ return null;
726
+ if (opts.requireSelfReview === false)
727
+ return null;
728
+ if (steps < SELF_REVIEW_MIN_STEPS)
729
+ return null;
730
+ if (progress.mutatedPaths.length === 0)
731
+ return null;
732
+ // The verification gate owns "did you check it"; self-review only runs once
733
+ // that question is settled, so the two can never both fire in one turn.
734
+ const activity = assessEditActivity(progress.successfulToolCalls, progress.verificationEvidence, progress.mutatedPaths);
735
+ if (activity.needsVerification)
736
+ return null;
737
+ return selfReviewNudge(currentAsk(opts));
738
+ };
649
739
  for (;;) {
650
740
  // ── Bounded auto-continuation on the STEP BOUND ────────────────────────
651
741
  // The model still wanted to act when the budget ran out (a long build, a
@@ -675,6 +765,19 @@ async function runToolLoopInner(opts, progress) {
675
765
  if (trimmedResult.trimmed > 0) {
676
766
  thread.length = 0;
677
767
  thread.push(...trimmedResult.thread);
768
+ // Keep a memory of what the turn DID after its raw outputs are gone:
769
+ // without this, a long turn loses the VERDICTS (did the build pass?) and
770
+ // re-runs work it already finished. Deterministic and bounded — the same
771
+ // philosophy as trimThreadBudget itself (facts, no summarizer, no drift).
772
+ if (isWorkDigestEnabled(context.configManager)) {
773
+ const digest = buildWorkDigest({
774
+ successfulTools: progress.successfulToolCalls,
775
+ mutatedPaths: progress.mutatedPaths,
776
+ executedActions: progress.executedActions,
777
+ });
778
+ if (digest)
779
+ upsertWorkDigest(thread, digest);
780
+ }
678
781
  deps.onEvent?.(` ✂️ ${trimmedResult.trimmed} old tool result(s) trimmed to fit the ${Math.round(budgetChars / 1000)}K-char context budget.`);
679
782
  }
680
783
  }
@@ -999,6 +1102,32 @@ async function runToolLoopInner(opts, progress) {
999
1102
  thread.push({ role: 'user', content: repeatNudge(closing, runTrace.priorAnswer(closing)) });
1000
1103
  continue;
1001
1104
  }
1105
+ // ── ZERO-ACTION gate (bounded, once) ────────────────────────────────
1106
+ // The request DIRECTED work on the workspace and the turn is ending
1107
+ // without having run a single tool: not a dropped promise, not a
1108
+ // permission question, not an authored file — just prose where the work
1109
+ // should be. The gates above all key on a positive shape in the reply and
1110
+ // so miss a plain non-answer; this one keys on the REQUEST and the
1111
+ // ABSENCE of action, which is exactly the shape that ended a fully
1112
+ // specified four-part coding ask with nothing done. Bounded once; the
1113
+ // residual is reported by `noActionTaken` below, whether or not the nudge
1114
+ // is enabled.
1115
+ if (actionNudges < 1 &&
1116
+ zeroActionGateApplies(opts, progress, schemas.length, requestText, authorization.authorized)) {
1117
+ actionNudges += 1;
1118
+ stepLimit += 1;
1119
+ deps.onEvent?.(' 🛠️ The request asked for work and nothing was done — telling the model to do it now.');
1120
+ traceEvent({
1121
+ kind: 'gate',
1122
+ gate: 'action',
1123
+ summary: 'the request directed work on the workspace and the turn performed none — one bounded nudge to carry it out',
1124
+ });
1125
+ // The non-answer must not become the delivered answer either way.
1126
+ lastContent = '';
1127
+ thread.push({ role: 'assistant', content: response.content });
1128
+ thread.push({ role: 'user', content: zeroActionNudge(currentAsk(opts)) });
1129
+ continue;
1130
+ }
1002
1131
  // S1 (both exits): the MOST SUBSTANTIVE content seen wins here too —
1003
1132
  // a short closing step ("Sent it to her! ✅") with no tool calls must
1004
1133
  // not clobber the deliverable (poem/essay) the model composed in an
@@ -1055,6 +1184,22 @@ async function runToolLoopInner(opts, progress) {
1055
1184
  });
1056
1185
  continue;
1057
1186
  }
1187
+ // SELF-REVIEW (no-tools exit) — a substantial, already-verified turn is
1188
+ // ending; spend ONE bounded pass to check the result against the ask.
1189
+ const review = selfReviewCorrection();
1190
+ if (review) {
1191
+ selfReviewNudges += 1;
1192
+ stepLimit += 1;
1193
+ deps.onEvent?.(' 🧭 Substantial turn ending — asking the model to review the result against the original ask.');
1194
+ traceEvent({
1195
+ kind: 'gate',
1196
+ gate: 'self-review',
1197
+ summary: 'a substantial turn that changed files reached its end — one bounded nudge to check the result against the original ask',
1198
+ });
1199
+ thread.push({ role: 'assistant', content: response.content });
1200
+ thread.push({ role: 'user', content: review });
1201
+ continue;
1202
+ }
1058
1203
  return {
1059
1204
  content: response.content.length >= lastContent.length ? response.content : lastContent,
1060
1205
  followups,
@@ -1323,6 +1468,7 @@ async function runToolLoopInner(opts, progress) {
1323
1468
  // What each call actually delivered (post hint/tip decoration), kept in
1324
1469
  // call order for the endsAgentStep exit below.
1325
1470
  const delivered = new Array(plans.length).fill('');
1471
+ let stepAnySuccess = false;
1326
1472
  for (let i = 0; i < plans.length; i += 1) {
1327
1473
  const call = plans[i].call;
1328
1474
  const rawResult = executed[i];
@@ -1342,8 +1488,10 @@ async function runToolLoopInner(opts, progress) {
1342
1488
  const ranOk = refusal === null &&
1343
1489
  !rawResult.startsWith('Error:') &&
1344
1490
  (!DELIVERY_TOOL_NAMES.has(call.name) || deliveryResultSucceeded(rawResult));
1345
- if (ranOk)
1491
+ if (ranOk) {
1346
1492
  progress.successfulToolCalls.push(call.name);
1493
+ stepAnySuccess = true;
1494
+ }
1347
1495
  if (call.name === 'gateway_send' && deliveryResultSucceeded(rawResult)) {
1348
1496
  progress.deliveryConfirmed = true;
1349
1497
  }
@@ -1367,6 +1515,17 @@ async function runToolLoopInner(opts, progress) {
1367
1515
  }
1368
1516
  }
1369
1517
  else if (refusal !== null || rawResult.startsWith('Error:')) {
1518
+ // SELF-DIAGNOSIS — record an action that RAN and FAILED (a non-zero
1519
+ // exit, a timeout, a tool error), keyed by the action itself (the
1520
+ // command / path). The gate below fires when the SAME action repeats:
1521
+ // a refusal was already remembered across runs (step-handoff), but a
1522
+ // command that RUNS and FAILS was not, so the live macOS-build turn
1523
+ // re-issued `npx tauri build` ~10 times against the same missing Cargo
1524
+ // and never once stopped to diagnose. Refusals are the other record
1525
+ // below; this one is for the calls that actually executed.
1526
+ if (refusal === null) {
1527
+ runTrace.recordFailure(call.name, failureActionOf(call), rawResult);
1528
+ }
1370
1529
  // Stage 2 — the other half of noticing a loop: the SAME refusal, twice.
1371
1530
  // The run records it with its reason, so a later step (or a self-report)
1372
1531
  // can see "blocked twice, identically" instead of rediscovering it — and
@@ -1443,6 +1602,52 @@ async function runToolLoopInner(opts, progress) {
1443
1602
  });
1444
1603
  thread.push({ role: 'tool', toolCallId: call.id, content: resultText });
1445
1604
  }
1605
+ // Track the generalized stall: a step that RAN tools but succeeded at none
1606
+ // of them extends the streak; any success clears it. A text-only step (no
1607
+ // tools) leaves it as it was — the model is thinking, not failing.
1608
+ if (plans.length > 0)
1609
+ failedToolSteps = stepAnySuccess ? 0 : failedToolSteps + 1;
1610
+ // ── SELF-DIAGNOSIS nudge (bounded, once) ──────────────────────────────
1611
+ // The loop's missing introspection: when the SAME action has failed more
1612
+ // than once, re-issuing it is a loop, not progress. This is the exact
1613
+ // shape of the live macOS-build turn — `npx tauri build` was retried ~10
1614
+ // times against the same missing Rust/Cargo, each attempt rediscovering
1615
+ // the same wall, and the run never stopped to ask WHY. Unlike the
1616
+ // end-of-turn gates (permission/repeat/deliverable), which fire only when
1617
+ // the model stops calling tools, this one must fire MID-TURN while the
1618
+ // model is still looping on the failing call — so it is injected here,
1619
+ // right after the step's results are in the thread, and the next model
1620
+ // step sees it. Bounded once per turn and keyed on the action so the same
1621
+ // repeat cannot trigger it twice.
1622
+ if (diagnosisNudges < 1) {
1623
+ const repeated = runTrace.repeatedFailure();
1624
+ if (repeated && repeated.times >= 2) {
1625
+ diagnosisNudges += 1;
1626
+ stepLimit += 1;
1627
+ deps.onEvent?.(` 🩺 Same action failed ${repeated.times}× — asking the model to diagnose the cause instead of retrying.`);
1628
+ traceEvent({
1629
+ kind: 'gate',
1630
+ gate: 'diagnosis',
1631
+ summary: `the same action failed ${repeated.times} times (${repeated.tool}: ${repeated.action.slice(0, 120)}) — ` +
1632
+ 'one bounded nudge to diagnose the root cause and change the approach',
1633
+ });
1634
+ thread.push({ role: 'user', content: repeatedFailureNudge(repeated) });
1635
+ }
1636
+ else if (failedToolSteps >= NO_PROGRESS_STALL_STEPS) {
1637
+ // Stage 2 — no SINGLE action repeated, but the last N steps each ran
1638
+ // tools and none succeeded. Same diagnosis demanded, different evidence.
1639
+ diagnosisNudges += 1;
1640
+ stepLimit += 1;
1641
+ deps.onEvent?.(` 🩺 ${failedToolSteps} steps with no successful action — asking the model to diagnose the stall.`);
1642
+ traceEvent({
1643
+ kind: 'gate',
1644
+ gate: 'diagnosis',
1645
+ summary: `${failedToolSteps} consecutive steps ran tools and none succeeded — ` +
1646
+ 'one bounded nudge to diagnose the shared cause and change the approach',
1647
+ });
1648
+ thread.push({ role: 'user', content: noProgressNudge(runTrace.recentFailures(3), failedToolSteps) });
1649
+ }
1650
+ }
1446
1651
  // Tiered exposure: after EVERY executed tool call, union any newly
1447
1652
  // loaded toolset schemas into the live set so the NEXT model step can
1448
1653
  // call them natively (tool_search load → loadedExtraTools → here).
@@ -1571,6 +1776,40 @@ async function runToolLoopInner(opts, progress) {
1571
1776
  thread.push({ role: 'user', content: deliverableNudge(authorization.requestedPath) });
1572
1777
  continue;
1573
1778
  }
1779
+ // ZERO-ACTION gate (concluding exit) — the same bounded pass as the
1780
+ // no-tools exit, for a turn that concluded (suggest_followups) without
1781
+ // ever touching the workspace a directed request asked it to change.
1782
+ if (actionNudges < 1 &&
1783
+ zeroActionGateApplies(opts, progress, schemas.length, requestText, authorization.authorized)) {
1784
+ actionNudges += 1;
1785
+ stepLimit += 1;
1786
+ deps.onEvent?.(' 🛠️ The request asked for work and nothing was done — telling the model to do it now.');
1787
+ traceEvent({
1788
+ kind: 'gate',
1789
+ gate: 'action',
1790
+ summary: 'the request directed work on the workspace and the turn performed none — one bounded nudge to carry it out',
1791
+ });
1792
+ lastContent = '';
1793
+ thread.push({ role: 'assistant', content: response.content });
1794
+ thread.push({ role: 'user', content: zeroActionNudge(currentAsk(opts)) });
1795
+ continue;
1796
+ }
1797
+ // SELF-REVIEW (concluding exit) — same bounded pass, placed AFTER the
1798
+ // verification/deliverable gates so those settle their questions first.
1799
+ const review = selfReviewCorrection();
1800
+ if (review) {
1801
+ selfReviewNudges += 1;
1802
+ stepLimit += 1;
1803
+ deps.onEvent?.(' 🧭 Substantial turn ending — asking the model to review the result against the original ask.');
1804
+ traceEvent({
1805
+ kind: 'gate',
1806
+ gate: 'self-review',
1807
+ summary: 'a substantial turn that changed files reached its end — one bounded nudge to check the result against the original ask',
1808
+ });
1809
+ thread.push({ role: 'assistant', content: response.content });
1810
+ thread.push({ role: 'user', content: review });
1811
+ continue;
1812
+ }
1574
1813
  const content = response.content.length >= lastContent.length ? response.content : lastContent;
1575
1814
  return {
1576
1815
  content,
@@ -1921,9 +2160,23 @@ export async function runToolLoop(opts) {
1921
2160
  if (!result.cancelled &&
1922
2161
  progress.mutatedPaths.length === 0 &&
1923
2162
  !replyAsksTheReader(result.content) &&
2163
+ // A request that forbade writes cannot have "failed to deliver" a file.
2164
+ !requestForbidsWrites(lastUserText(opts.messages)) &&
1924
2165
  wantsAuthoredArtifact(lastUserText(opts.messages))) {
1925
2166
  result.undeliveredArtifact = true;
1926
2167
  }
2168
+ // ZERO-ACTION honesty — a request that DIRECTED work on the workspace was
2169
+ // answered with nothing at all: no tool succeeded, nothing was written. The
2170
+ // flag is a function of what the turn DID, never of configuration, so a
2171
+ // caller can never read "completed" from a turn whose request it never
2172
+ // touched. Authored-artifact asks are left to `undeliveredArtifact` above so
2173
+ // one turn is never reported under two names.
2174
+ const askText = lastUserText(opts.messages);
2175
+ if (!hasProductiveAction(progress) &&
2176
+ progress.mutatedPaths.length === 0 &&
2177
+ requestRequiresWorkspaceAction(askText, requestAuthorizesWrites(askText).authorized)) {
2178
+ result.noActionTaken = true;
2179
+ }
1927
2180
  }
1928
2181
  return result;
1929
2182
  }
@@ -1961,8 +2214,109 @@ function deliverableGateApplies(opts, content, progress, schemaCount, requestTex
1961
2214
  return false;
1962
2215
  if (replyAsksTheReader(content))
1963
2216
  return false;
2217
+ // A negative instruction outranks every positive signal. "Do not write any
2218
+ // files — answer in chat" contains a creation verb and a file-shaped noun, so
2219
+ // without this the gate read it as an authored-artifact ask and wrote a file
2220
+ // against the user's explicit instruction.
2221
+ if (requestForbidsWrites(requestText))
2222
+ return false;
1964
2223
  return wantsAuthoredArtifact(requestText);
1965
2224
  }
2225
+ /**
2226
+ * Tools that do not count as having DONE anything on their own. `suggest_followups`
2227
+ * concludes a turn; it produces no work, so a turn that only called it has still
2228
+ * performed nothing the request asked for. Every other successful tool counts as
2229
+ * an action (a read is an action), which keeps the zero-action gate conservative.
2230
+ */
2231
+ /**
2232
+ * A token that NAMES a file (a real source/config extension). The strongest
2233
+ * signal that the request is about the WORKSPACE, not a chat answer.
2234
+ */
2235
+ const REQUEST_FILE_TOKEN_RE = /\b[\w./~-]+\.(?:js|mjs|cjs|ts|tsx|jsx|py|rb|go|rs|java|kt|cs|cpp|cxx|cc|c|h|hpp|json|ya?ml|toml|ini|cfg|conf|md|markdown|txt|csv|tsv|html?|css|scss|sass|less|sql|sh|bash|zsh|fish|env|lock|xml|gradle|properties|vue|svelte|php|lua|pl|swift|dart|scala)\b/i;
2236
+ /** A verb that asks for a change to the workspace. */
2237
+ const WORK_EDIT_VERB_RE = /\b(?:fix|repair|refactor|edit|modify|updat|chang|add|implement|creat|writ|delet|remov|renam|rewrit|migrat|correct|debug|patch|tweak|scaffold)\w*/i;
2238
+ /** A code/workspace noun that pairs with the verb above. */
2239
+ const CODE_NOUN_RE = /\b(?:file|files|function|functions|method|methods|class|classes|module|modules|script|scripts|component|components|test|tests|suite|api|endpoint|endpoints|route|routes|schema|schemas|migration|migrations|package|dependency|dependencies|import|imports|config|configuration|type|types|interface|interfaces|bug|bugs|repo|repository|codebase|project|source)\b/i;
2240
+ /**
2241
+ * Does this request DIRECT work on the workspace — the precondition for the
2242
+ * ZERO-ACTION gate?
2243
+ *
2244
+ * Deliberately narrower than "the request authorizes writes": the gate spends a
2245
+ * model step, so it must only fire on an ask that genuinely needs the tools.
2246
+ * `requestAuthorizesWrites` is true for a prose deliverable ("draft an
2247
+ * itinerary") or a review ("assess this project"), which are correctly answered
2248
+ * in chat — nudging those would turn one turn into two for no reason (found by
2249
+ * the existing suite). The workspace signal is therefore explicit: the request
2250
+ * either names a FILE, or pairs an edit verb with a code noun.
2251
+ */
2252
+ function requestRequiresWorkspaceAction(requestText, authorized) {
2253
+ const text = (requestText || '').trim();
2254
+ if (!text)
2255
+ return false;
2256
+ if (!authorized)
2257
+ return false;
2258
+ if (requestForbidsWrites(text))
2259
+ return false;
2260
+ if (wantsAuthoredArtifact(text))
2261
+ return false;
2262
+ if (REQUEST_FILE_TOKEN_RE.test(text))
2263
+ return true;
2264
+ return WORK_EDIT_VERB_RE.test(text) && CODE_NOUN_RE.test(text);
2265
+ }
2266
+ const NON_PRODUCTIVE_TOOLS = new Set(['suggest_followups']);
2267
+ /** True when the turn performed at least one action beyond merely concluding. */
2268
+ function hasProductiveAction(progress) {
2269
+ return progress.successfulToolCalls.some((name) => !NON_PRODUCTIVE_TOOLS.has(name));
2270
+ }
2271
+ /**
2272
+ * Does the ZERO-ACTION gate apply to this turn?
2273
+ *
2274
+ * Every term is an independent, checkable fact — the conjunction is what keeps
2275
+ * the gate from firing on turns that are legitimately answer-only:
2276
+ *
2277
+ * - the request AUTHORIZES writes (`authorized`) — a create/maintenance ask,
2278
+ * not a question (`requestAuthorizesWrites` already vetoes pure questions);
2279
+ * - NO tool call succeeded this turn (a turn that gathered context and then
2280
+ * answered is out of scope — it did something, even if it then stalled);
2281
+ * - NOTHING was written (`mutatedPaths` is the loop's own proof a write
2282
+ * landed, so a heredoc through `run_terminal` is not nudged);
2283
+ * - tools are actually available (a caller that exposed none cannot comply);
2284
+ * - the request did NOT forbid writes (a negative instruction outranks every
2285
+ * positive signal — see `requestForbidsWrites`);
2286
+ * - it is NOT an authored-artifact ask: `wantsAuthoredArtifact` requests are
2287
+ * the DELIVERABLE gate's job, with their own narrower message, and running
2288
+ * both would spend two nudges on one ask.
2289
+ */
2290
+ function zeroActionGateApplies(opts, progress, schemaCount, requestText, authorized) {
2291
+ if (opts.requireAction === false)
2292
+ return false;
2293
+ if (schemaCount === 0)
2294
+ return false;
2295
+ if (hasProductiveAction(progress))
2296
+ return false;
2297
+ if (progress.mutatedPaths.length > 0)
2298
+ return false;
2299
+ return requestRequiresWorkspaceAction(requestText, authorized);
2300
+ }
2301
+ /**
2302
+ * The bounded ZERO-ACTION correction — the loop telling the model, in one step,
2303
+ * that a directed request has not been touched yet.
2304
+ *
2305
+ * Names the ask (so the model cannot claim it did not know what was wanted),
2306
+ * states plainly that nothing ran and nothing changed, and closes the two escape
2307
+ * hatches that produced the observed non-actions: do not re-ask a request that is
2308
+ * already specified, and do not answer a work request with a plan, an apology or
2309
+ * a question. It still leaves a REAL blocker as a legitimate way out — the goal
2310
+ * is the work, not compliance theatre.
2311
+ */
2312
+ export function zeroActionNudge(ask) {
2313
+ const quoted = (ask || '').trim().replace(/\s+/g, ' ').slice(0, 300);
2314
+ return ('Nothing has been done yet: you have not called a single tool this turn, so no file was changed and nothing was checked.' +
2315
+ (quoted ? ` The request already asked for this work: "${quoted}".` : '') +
2316
+ '\nYou have the tools to do it — read what you need, make the change, and run the check NOW.' +
2317
+ ' Do NOT reply with a plan, a summary of what you would do, an apology, or a request for the user to restate or confirm' +
2318
+ ' a request that is already complete. If something genuinely blocks you, say exactly what it is and why; otherwise do the work.');
2319
+ }
1966
2320
  /**
1967
2321
  * The text of the LAST user message in the thread — what the user is actually
1968
2322
  * asking for right now, as opposed to the history above it. Used to derive
@@ -2148,4 +2502,92 @@ export function trimThreadBudget(thread, maxChars = DEFAULT_THREAD_BUDGET_CHARS)
2148
2502
  }
2149
2503
  return { thread: out, trimmed };
2150
2504
  }
2505
+ // ─── Within-turn work digest (the memory a trimmed thread keeps) ────────────
2506
+ // `trimThreadBudget` keeps the first 500 chars of each old tool result, but past
2507
+ // that the model loses the VERDICT of what it ran and can re-run finished work.
2508
+ // `working-state` solves this ACROSS turns; this solves it WITHIN one. It is
2509
+ // deliberately deterministic and LLM-free (facts: what changed, what commands
2510
+ // ran and whether they passed, which tools were used) — no summarizer, no
2511
+ // latency, no drift, exactly like the ledger and the budget it complements.
2512
+ /** Marker prefix identifying the loop's within-turn work digest message. */
2513
+ export const WORK_DIGEST_MARKER = '[work digest — your actions so far this turn]';
2514
+ /**
2515
+ * Format the turn's actions so far as a bounded, model-readable digest. Returns
2516
+ * '' when there is nothing worth saying, so a pristine turn adds no noise.
2517
+ */
2518
+ export function buildWorkDigest(input) {
2519
+ const lines = [];
2520
+ const changed = [...new Set(input.mutatedPaths.filter(Boolean))];
2521
+ if (changed.length > 0) {
2522
+ const shown = changed.slice(-12);
2523
+ lines.push(`• Files changed (${changed.length}): ${shown.join(', ')}${changed.length > shown.length ? ', …' : ''}`);
2524
+ }
2525
+ // Commands with their verdict, newest first, deduped by command+verdict — the
2526
+ // single most valuable thing to retain (did the build/test pass?).
2527
+ const cmds = input.executedActions.filter((a) => typeof a.command === 'string' && a.command.trim());
2528
+ if (cmds.length > 0) {
2529
+ const seen = new Set();
2530
+ const shown = [];
2531
+ for (let i = cmds.length - 1; i >= 0 && shown.length < 8; i -= 1) {
2532
+ const a = cmds[i];
2533
+ const key = `${a.ok ? 'ok' : 'fail'}:${a.command}`;
2534
+ if (seen.has(key))
2535
+ continue;
2536
+ seen.add(key);
2537
+ shown.unshift(`${a.ok ? '✅' : '❌'} ${a.command}`);
2538
+ }
2539
+ lines.push('• Commands run:');
2540
+ for (const s of shown)
2541
+ lines.push(` ${s}`);
2542
+ }
2543
+ const tools = [...new Set(input.successfulTools)];
2544
+ if (tools.length > 0)
2545
+ lines.push(`• Tools used: ${tools.join(', ')}`);
2546
+ if (lines.length === 0)
2547
+ return '';
2548
+ return `${WORK_DIGEST_MARKER}\n${lines.join('\n')}`;
2549
+ }
2550
+ /**
2551
+ * Insert or refresh the single work-digest message. Idempotent: a digest already
2552
+ * in the thread is UPDATED in place, never stacked, so repeated compaction in a
2553
+ * long turn keeps one current digest rather than accumulating stale ones. Placed
2554
+ * just after the system prompt + first user message so it sits with the ask and
2555
+ * survives the NEXT compaction (the tail is never trimmed).
2556
+ */
2557
+ export function upsertWorkDigest(thread, digest) {
2558
+ const existing = thread.findIndex((m) => m.content.startsWith(WORK_DIGEST_MARKER));
2559
+ if (existing !== -1) {
2560
+ thread[existing] = { ...thread[existing], content: digest };
2561
+ return;
2562
+ }
2563
+ let at = 0;
2564
+ while (at < thread.length && thread[at].role === 'system')
2565
+ at += 1;
2566
+ const firstUser = thread.findIndex((m) => m.role === 'user');
2567
+ const insertAt = Math.min(thread.length, firstUser === -1 ? at : Math.max(at, firstUser + 1));
2568
+ thread.splice(insertAt, 0, { role: 'user', content: digest });
2569
+ }
2570
+ /**
2571
+ * The bounded SELF-REVIEW correction — the loop's stand-in for a reviewer who
2572
+ * asks "is this actually what was asked for?".
2573
+ *
2574
+ * Fired once, only for a substantial turn that changed files and has already
2575
+ * been verified (the verification gate owns "did you check"). Its job is
2576
+ * orthogonal: catch a result that is verified but does not satisfy the WHOLE
2577
+ * original ask. It demands evidence from THIS turn and forbids padding the
2578
+ * answer with more prose — the failure mode it targets is a confident, verified
2579
+ * answer to a slightly wrong question.
2580
+ */
2581
+ export function selfReviewNudge(ask) {
2582
+ const quoted = (ask || '').trim().replace(/\s+/g, ' ').slice(0, 300);
2583
+ return ('Before you finish, REVIEW your result against the ORIGINAL request' +
2584
+ (quoted ? `: "${quoted}".` : '.') +
2585
+ '\nAnswer these to yourself in one short pass, and fix anything that is not true:' +
2586
+ '\n 1. Does what you produced satisfy EVERY part of that request — not just the part you found easiest?' +
2587
+ '\n 2. Is each claim in your answer backed by a tool result from THIS turn — or is it an assumption you did not check?' +
2588
+ '\n 3. Is anything the request asked for still missing, half-done, or done for the wrong target?' +
2589
+ '\nIf everything is satisfied and evidenced, reply with a brief confirmation and stop.' +
2590
+ ' If something is missing, do it NOW — or state plainly and specifically what is not done and why.' +
2591
+ ' Do NOT restate the work in more words — verify it.');
2592
+ }
2151
2593
  //# sourceMappingURL=tool-loop.js.map