agent-nuvira 3.3.11 → 3.3.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (220) hide show
  1. package/README.md +32 -1
  2. package/dist/agents/agents/reasoner.d.ts.map +1 -1
  3. package/dist/agents/agents/reasoner.js +13 -0
  4. package/dist/agents/agents/reasoner.js.map +1 -1
  5. package/dist/agents/orchestrator.d.ts.map +1 -1
  6. package/dist/agents/orchestrator.js +56 -20
  7. package/dist/agents/orchestrator.js.map +1 -1
  8. package/dist/cli/chat.d.ts +36 -0
  9. package/dist/cli/chat.d.ts.map +1 -1
  10. package/dist/cli/chat.js +456 -43
  11. package/dist/cli/chat.js.map +1 -1
  12. package/dist/cli/cli-program.d.ts.map +1 -1
  13. package/dist/cli/cli-program.js +8 -0
  14. package/dist/cli/cli-program.js.map +1 -1
  15. package/dist/cli/config.d.ts.map +1 -1
  16. package/dist/cli/config.js +35 -4
  17. package/dist/cli/config.js.map +1 -1
  18. package/dist/cli/execute.d.ts.map +1 -1
  19. package/dist/cli/execute.js +14 -0
  20. package/dist/cli/execute.js.map +1 -1
  21. package/dist/cli/knowledge.d.ts +23 -0
  22. package/dist/cli/knowledge.d.ts.map +1 -0
  23. package/dist/cli/knowledge.js +132 -0
  24. package/dist/cli/knowledge.js.map +1 -0
  25. package/dist/cli/loop-executor.d.ts.map +1 -1
  26. package/dist/cli/loop-executor.js +57 -13
  27. package/dist/cli/loop-executor.js.map +1 -1
  28. package/dist/cli/omniroute-control.d.ts +88 -0
  29. package/dist/cli/omniroute-control.d.ts.map +1 -0
  30. package/dist/cli/omniroute-control.js +179 -0
  31. package/dist/cli/omniroute-control.js.map +1 -0
  32. package/dist/cli/omniroute.d.ts +21 -0
  33. package/dist/cli/omniroute.d.ts.map +1 -0
  34. package/dist/cli/omniroute.js +67 -0
  35. package/dist/cli/omniroute.js.map +1 -0
  36. package/dist/cli/tool-install-prompt.d.ts +58 -0
  37. package/dist/cli/tool-install-prompt.d.ts.map +1 -1
  38. package/dist/cli/tool-install-prompt.js +137 -0
  39. package/dist/cli/tool-install-prompt.js.map +1 -1
  40. package/dist/cli/weak-model-prompt.d.ts +16 -7
  41. package/dist/cli/weak-model-prompt.d.ts.map +1 -1
  42. package/dist/cli/weak-model-prompt.js +24 -8
  43. package/dist/cli/weak-model-prompt.js.map +1 -1
  44. package/dist/config/types.d.ts +44 -0
  45. package/dist/config/types.d.ts.map +1 -1
  46. package/dist/context/cache.d.ts +35 -1
  47. package/dist/context/cache.d.ts.map +1 -1
  48. package/dist/context/cache.js +31 -2
  49. package/dist/context/cache.js.map +1 -1
  50. package/dist/context/history.d.ts +21 -0
  51. package/dist/context/history.d.ts.map +1 -1
  52. package/dist/context/history.js +50 -0
  53. package/dist/context/history.js.map +1 -1
  54. package/dist/gateway/hook-contract.d.ts +173 -0
  55. package/dist/gateway/hook-contract.d.ts.map +1 -0
  56. package/dist/gateway/hook-contract.js +352 -0
  57. package/dist/gateway/hook-contract.js.map +1 -0
  58. package/dist/gateway/hooks.d.ts.map +1 -1
  59. package/dist/gateway/hooks.js +14 -0
  60. package/dist/gateway/hooks.js.map +1 -1
  61. package/dist/inference/interface.d.ts +17 -0
  62. package/dist/inference/interface.d.ts.map +1 -1
  63. package/dist/inference/model-catalog.d.ts +17 -0
  64. package/dist/inference/model-catalog.d.ts.map +1 -1
  65. package/dist/inference/model-catalog.js +42 -7
  66. package/dist/inference/model-catalog.js.map +1 -1
  67. package/dist/inference/model-id-validation.d.ts +57 -0
  68. package/dist/inference/model-id-validation.d.ts.map +1 -0
  69. package/dist/inference/model-id-validation.js +127 -0
  70. package/dist/inference/model-id-validation.js.map +1 -0
  71. package/dist/inference/openai-compat-adapter.d.ts.map +1 -1
  72. package/dist/inference/openai-compat-adapter.js +12 -1
  73. package/dist/inference/openai-compat-adapter.js.map +1 -1
  74. package/dist/inference/provider-catalog.d.ts +11 -0
  75. package/dist/inference/provider-catalog.d.ts.map +1 -1
  76. package/dist/inference/provider-catalog.js +80 -0
  77. package/dist/inference/provider-catalog.js.map +1 -1
  78. package/dist/inference/route-resolver.d.ts +2 -8
  79. package/dist/inference/route-resolver.d.ts.map +1 -1
  80. package/dist/inference/route-resolver.js +17 -0
  81. package/dist/inference/route-resolver.js.map +1 -1
  82. package/dist/inference/tool-call-utils.d.ts +4 -3
  83. package/dist/inference/tool-call-utils.d.ts.map +1 -1
  84. package/dist/inference/tool-call-utils.js +26 -12
  85. package/dist/inference/tool-call-utils.js.map +1 -1
  86. package/dist/inference/tools.d.ts +1 -0
  87. package/dist/inference/tools.d.ts.map +1 -1
  88. package/dist/inference/tools.js +20 -3
  89. package/dist/inference/tools.js.map +1 -1
  90. package/dist/learning/agentic-route-gate.d.ts +104 -0
  91. package/dist/learning/agentic-route-gate.d.ts.map +1 -0
  92. package/dist/learning/agentic-route-gate.js +125 -0
  93. package/dist/learning/agentic-route-gate.js.map +1 -0
  94. package/dist/learning/auto-router.d.ts +89 -0
  95. package/dist/learning/auto-router.d.ts.map +1 -1
  96. package/dist/learning/auto-router.js +188 -8
  97. package/dist/learning/auto-router.js.map +1 -1
  98. package/dist/learning/build-prerequisites.d.ts +68 -0
  99. package/dist/learning/build-prerequisites.d.ts.map +1 -0
  100. package/dist/learning/build-prerequisites.js +267 -0
  101. package/dist/learning/build-prerequisites.js.map +1 -0
  102. package/dist/learning/continuation-intent.d.ts +43 -0
  103. package/dist/learning/continuation-intent.d.ts.map +1 -0
  104. package/dist/learning/continuation-intent.js +82 -0
  105. package/dist/learning/continuation-intent.js.map +1 -0
  106. package/dist/learning/error-repair.d.ts +23 -1
  107. package/dist/learning/error-repair.d.ts.map +1 -1
  108. package/dist/learning/error-repair.js +58 -0
  109. package/dist/learning/error-repair.js.map +1 -1
  110. package/dist/learning/knowledge-base.d.ts +124 -0
  111. package/dist/learning/knowledge-base.d.ts.map +1 -0
  112. package/dist/learning/knowledge-base.js +337 -0
  113. package/dist/learning/knowledge-base.js.map +1 -0
  114. package/dist/learning/model-harness.d.ts +19 -0
  115. package/dist/learning/model-harness.d.ts.map +1 -1
  116. package/dist/learning/model-harness.js +27 -0
  117. package/dist/learning/model-harness.js.map +1 -1
  118. package/dist/learning/model-selection.d.ts +2 -1
  119. package/dist/learning/model-selection.d.ts.map +1 -1
  120. package/dist/learning/model-selection.js +35 -2
  121. package/dist/learning/model-selection.js.map +1 -1
  122. package/dist/learning/project-docs.d.ts +83 -0
  123. package/dist/learning/project-docs.d.ts.map +1 -0
  124. package/dist/learning/project-docs.js +132 -0
  125. package/dist/learning/project-docs.js.map +1 -0
  126. package/dist/learning/prompt-budget.d.ts +91 -0
  127. package/dist/learning/prompt-budget.d.ts.map +1 -0
  128. package/dist/learning/prompt-budget.js +99 -0
  129. package/dist/learning/prompt-budget.js.map +1 -0
  130. package/dist/learning/quota-ledger.d.ts +13 -0
  131. package/dist/learning/quota-ledger.d.ts.map +1 -1
  132. package/dist/learning/quota-ledger.js +28 -0
  133. package/dist/learning/quota-ledger.js.map +1 -1
  134. package/dist/learning/reasoning-trace.d.ts +55 -2
  135. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  136. package/dist/learning/reasoning-trace.js +48 -0
  137. package/dist/learning/reasoning-trace.js.map +1 -1
  138. package/dist/learning/resilient-call.d.ts.map +1 -1
  139. package/dist/learning/resilient-call.js +16 -3
  140. package/dist/learning/resilient-call.js.map +1 -1
  141. package/dist/learning/routing-history.d.ts +13 -0
  142. package/dist/learning/routing-history.d.ts.map +1 -1
  143. package/dist/learning/routing-history.js.map +1 -1
  144. package/dist/learning/seeded-benchmark.d.ts +14 -0
  145. package/dist/learning/seeded-benchmark.d.ts.map +1 -1
  146. package/dist/learning/seeded-benchmark.js +43 -4
  147. package/dist/learning/seeded-benchmark.js.map +1 -1
  148. package/dist/learning/turn-report.d.ts +78 -0
  149. package/dist/learning/turn-report.d.ts.map +1 -0
  150. package/dist/learning/turn-report.js +140 -0
  151. package/dist/learning/turn-report.js.map +1 -0
  152. package/dist/security/secret-scan.d.ts +156 -0
  153. package/dist/security/secret-scan.d.ts.map +1 -0
  154. package/dist/security/secret-scan.js +525 -0
  155. package/dist/security/secret-scan.js.map +1 -0
  156. package/dist/tools/edit-verification.d.ts +2 -11
  157. package/dist/tools/edit-verification.d.ts.map +1 -1
  158. package/dist/tools/edit-verification.js +33 -1
  159. package/dist/tools/edit-verification.js.map +1 -1
  160. package/dist/tools/loop-skill-hint.d.ts +51 -0
  161. package/dist/tools/loop-skill-hint.d.ts.map +1 -1
  162. package/dist/tools/loop-skill-hint.js +87 -2
  163. package/dist/tools/loop-skill-hint.js.map +1 -1
  164. package/dist/tools/plan-store.d.ts +126 -3
  165. package/dist/tools/plan-store.d.ts.map +1 -1
  166. package/dist/tools/plan-store.js +290 -6
  167. package/dist/tools/plan-store.js.map +1 -1
  168. package/dist/tools/registry.d.ts +20 -1
  169. package/dist/tools/registry.d.ts.map +1 -1
  170. package/dist/tools/registry.js +165 -4
  171. package/dist/tools/registry.js.map +1 -1
  172. package/dist/tools/remediation-ladder.d.ts +91 -0
  173. package/dist/tools/remediation-ladder.d.ts.map +1 -0
  174. package/dist/tools/remediation-ladder.js +308 -0
  175. package/dist/tools/remediation-ladder.js.map +1 -0
  176. package/dist/tools/run-terminal.d.ts.map +1 -1
  177. package/dist/tools/run-terminal.js +60 -2
  178. package/dist/tools/run-terminal.js.map +1 -1
  179. package/dist/tools/skills-hub.d.ts +7 -0
  180. package/dist/tools/skills-hub.d.ts.map +1 -1
  181. package/dist/tools/skills-hub.js +11 -2
  182. package/dist/tools/skills-hub.js.map +1 -1
  183. package/dist/tools/step-artifact.d.ts +35 -0
  184. package/dist/tools/step-artifact.d.ts.map +1 -0
  185. package/dist/tools/step-artifact.js +154 -0
  186. package/dist/tools/step-artifact.js.map +1 -0
  187. package/dist/tools/tool-loop.d.ts +35 -0
  188. package/dist/tools/tool-loop.d.ts.map +1 -1
  189. package/dist/tools/tool-loop.js +284 -9
  190. package/dist/tools/tool-loop.js.map +1 -1
  191. package/dist/tools/toolsets.js +4 -4
  192. package/dist/tools/toolsets.js.map +1 -1
  193. package/dist/tools/worktree.d.ts +11 -2
  194. package/dist/tools/worktree.d.ts.map +1 -1
  195. package/dist/tools/worktree.js +11 -2
  196. package/dist/tools/worktree.js.map +1 -1
  197. package/dist/utils/effect-verification.js +1 -1
  198. package/dist/utils/effect-verification.js.map +1 -1
  199. package/dist/web-dashboard/chat-console.d.ts +21 -1
  200. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  201. package/dist/web-dashboard/chat-console.js +33 -5
  202. package/dist/web-dashboard/chat-console.js.map +1 -1
  203. package/dist/web-dashboard/server.d.ts +29 -0
  204. package/dist/web-dashboard/server.d.ts.map +1 -1
  205. package/dist/web-dashboard/server.js +410 -5
  206. package/dist/web-dashboard/server.js.map +1 -1
  207. package/dist/web-dashboard/src/types.d.ts +137 -0
  208. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  209. package/dist/web-dashboard/workspace-guard.d.ts +7 -0
  210. package/dist/web-dashboard/workspace-guard.d.ts.map +1 -1
  211. package/dist/web-dashboard/workspace-guard.js +70 -0
  212. package/dist/web-dashboard/workspace-guard.js.map +1 -1
  213. package/package.json +4 -1
  214. package/src/web-dashboard/public/assets/index-DhiuR4S2.js +210 -0
  215. package/src/web-dashboard/public/assets/index-DhiuR4S2.js.map +1 -0
  216. package/src/web-dashboard/public/assets/index-tK8Y2qHB.css +1 -0
  217. package/src/web-dashboard/public/index.html +2 -2
  218. package/src/web-dashboard/public/assets/index-BdKf5Xw2.js +0 -207
  219. package/src/web-dashboard/public/assets/index-BdKf5Xw2.js.map +0 -1
  220. package/src/web-dashboard/public/assets/index-C9eBskc9.css +0 -1
@@ -49,6 +49,8 @@ import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, veri
49
49
  import { effectiveToolJsonSchemas, coreToolJsonSchemas, isToolEnabled, toolsetForTool } from './toolsets.js';
50
50
  import { deliverablesNamedIn, recordStepHandoff } from '../agents/step-handoff.js';
51
51
  import { fenceUntrustedToolOutput } from './untrusted-content.js';
52
+ import { installableToolsFromFailure, resolveInstallCommand, toolTakeoverInstruction, } from '../cli/tool-install-prompt.js';
53
+ import { matchPrerequisiteSignatures, prerequisiteTakeoverInstruction, checkProjectPrerequisites, createNodePrereqFs, formatPreflightFindings, } from '../learning/build-prerequisites.js';
52
54
  /**
53
55
  * Tools that MUTATE the workspace. A refusal of one of these is the only
54
56
  * refusal that leaves work undone — a declined `read_file` costs a step, a
@@ -61,6 +63,19 @@ const MUTATING_TOOL_NAMES = new Set([
61
63
  'run_terminal',
62
64
  'run_cli',
63
65
  ]);
66
+ /**
67
+ * The subset of mutating tools the PLAN gate HARD-BLOCKS on the first call —
68
+ * the ones that change the workspace's FILES. `run_terminal` / `run_cli` are
69
+ * deliberately excluded from the BLOCK (they still trigger the nudge): a
70
+ * terminal command is as often a test, a build or an inspection as it is an
71
+ * edit, and refusing it outright would stop the very step a plan is meant to
72
+ * reach. The nudge still tells the model to plan first.
73
+ */
74
+ const PLAN_GATED_TOOL_NAMES = new Set([
75
+ 'write_file',
76
+ 'edit_file',
77
+ 'propose_change',
78
+ ]);
64
79
  /** The path a mutating call was aimed at, when it named one. */
65
80
  function mutatedPathOf(args) {
66
81
  const a = args;
@@ -438,6 +453,15 @@ export const NO_PROGRESS_STALL_STEPS = 4;
438
453
  * where an extra pass earns its latency.
439
454
  */
440
455
  export const SELF_REVIEW_MIN_STEPS = 6;
456
+ /**
457
+ * S5 — how many `plan_todo` UPDATE calls one turn may make. The guard refuses
458
+ * repeated CREATES (the observed planner loop) but must allow updates, because
459
+ * an update is how the user's checklist advances; this cap only stops a model
460
+ * that replaces doing the work with spamming status changes. It sits well above
461
+ * a genuine multi-step job (the largest real plans in the tree are single
462
+ * digits) and far below a loop.
463
+ */
464
+ export const PLAN_TODO_UPDATE_CAP = 64;
441
465
  /** Default continuations granted per turn when the option is omitted. */
442
466
  export const DEFAULT_MAX_CONTINUATIONS = 2;
443
467
  /** Default extra steps granted per continuation. */
@@ -537,6 +561,12 @@ async function runToolLoopInner(opts, progress) {
537
561
  const maxParallelReads = Math.max(1, opts.maxParallelReads ?? MAX_PARALLEL_READS);
538
562
  const followups = [];
539
563
  const toolCallsRun = [];
564
+ // S5 — the plan_todo loop guard is split by ACTION (see the guard below):
565
+ // repeated CREATES were the observed planner loop; UPDATES are the tracking
566
+ // that keeps the user's checklist honest. Counted across the whole turn, like
567
+ // `toolCallsRun`, because a multi-step job runs many steps in one turn.
568
+ let planTodoCreates = 0;
569
+ let planTodoUpdates = 0;
540
570
  // Every collected suggestion passes through the shared normalizer, so the
541
571
  // loop's output is ALWAYS clean + structured (1–3 items, deduped, no leaked
542
572
  // tool JSON, capped prompt/label) regardless of what the model emitted —
@@ -715,6 +745,18 @@ async function runToolLoopInner(opts, progress) {
715
745
  // with an empty/bounded response (the "where is the essay?" bug).
716
746
  let lastContent = '';
717
747
  let bounded = false;
748
+ // D2.2 — has this turn already pre-flighted the project's build prerequisites?
749
+ // Checked once, before the FIRST build command the turn plans to run.
750
+ let prerequisitesPreflighted = false;
751
+ // E2 — has this turn already spent its one bounded "declare a plan first"
752
+ // nudge? Bounded once, like every other gate.
753
+ let planNudged = false;
754
+ // E2 (hard) — has this turn already spent its one bounded BLOCK of the first
755
+ // mutation? The plan gate is a hard requirement the first time it fires and a
756
+ // pure nudge after that: the first mutating batch is refused (see the gate
757
+ // below), and the second attempt runs even without a plan — a bounded block,
758
+ // never a wall.
759
+ let planBlocked = false;
718
760
  /**
719
761
  * Should the SELF-REVIEW gate fire now? Returns the correction to inject, or
720
762
  * null. Extracted so the two end-of-turn exits ask the question identically —
@@ -937,7 +979,12 @@ async function runToolLoopInner(opts, progress) {
937
979
  // isThinkOnlyResponse: continue instead of ending).
938
980
  if (deps.isThinkOnly ? deps.isThinkOnly(response.content) : isThinkOnlyResponse(response.content)) {
939
981
  // Feed an empty assistant step so the model continues in-context.
940
- thread.push({ role: 'assistant', content: response.content });
982
+ thread.push({
983
+ role: 'assistant',
984
+ content: response.content,
985
+ // Keep the thinking model's reasoning with the turn it belongs to.
986
+ ...(response.reasoningContent ? { reasoningContent: response.reasoningContent } : {}),
987
+ });
941
988
  thinkContinues += 1;
942
989
  // AN EMPTY RESPONSE IS A FAILURE, NOT REASONING. `isThinkOnlyResponse('')`
943
990
  // returns true, so a provider returning nothing (the loop arm falls over
@@ -1224,20 +1271,67 @@ async function runToolLoopInner(opts, progress) {
1224
1271
  arguments: JSON.stringify(tc.arguments),
1225
1272
  ...(tc.providerMeta ? { providerMeta: tc.providerMeta } : {}),
1226
1273
  })),
1274
+ // Thinking-mode reasoning is echoed back with the assistant turn that
1275
+ // carried these calls — DeepSeek v4 rejects the next request without it.
1276
+ ...(response.reasoningContent ? { reasoningContent: response.reasoningContent } : {}),
1227
1277
  });
1228
1278
  let endedAfterConcluding = false;
1279
+ // ── Provider-protocol safety: nudges decided while PLANNING must queue ──
1280
+ // An assistant message that carries `tool_calls` MUST be followed
1281
+ // immediately by a `tool` result for EVERY call id, before any other role.
1282
+ // A strict OpenAI-compatible API (DeepSeek native) rejects the NEXT request
1283
+ // otherwise: "An assistant message with 'tool_calls' must be followed by
1284
+ // tool messages responding to each 'tool_call_id'." The pre-flight and plan
1285
+ // gates decide during planning — before the results exist — so they push
1286
+ // their nudge into this queue and it is flushed after the results, keeping
1287
+ // the assistant→tool group intact. (Without this, a pinned/strict run died
1288
+ // on the third model call and looked like a provider fault.)
1289
+ const deferredUserNudges = [];
1229
1290
  const plans = toolCalls.map((call) => {
1230
1291
  const tool = getTool(call.name);
1231
1292
  const priorSameTool = toolCallsRun.filter((t) => t === call.name).length;
1232
1293
  toolCallsRun.push(call.name);
1233
- // S5 — PLANNER LOOP GUARD: allow the first plan of the turn, refuse
1234
- // repeats (the 6x planner loop observed in
1235
- // trace-1788059239352-k7zl03: 15.6K tokens, 2m25s, FAILED).
1236
- if (call.name === 'plan_todo' && priorSameTool >= 1) {
1237
- return {
1238
- call,
1239
- refuse: 'Error: plan_todo already called. You have a plan — now execute it. Do NOT call plan_todo again. Use write_file, run_terminal, or other execution tools to complete the work.',
1240
- };
1294
+ // S5 — PLANNER LOOP GUARD, now split by ACTION.
1295
+ //
1296
+ // The guard exists because of a real failure: the model called plan_todo
1297
+ // six times in one turn (trace-1788059239352-k7zl03 — 15.6K tokens,
1298
+ // 2m25s, FAILED) instead of doing the work. But `create` and `update` are
1299
+ // the SAME tool, and the original `priorSameTool >= 1` check refused BOTH
1300
+ // — so a plan could be declared once and then never advanced, and the
1301
+ // checklist the user watches went stale the moment the first step closed.
1302
+ // Only repeated CREATES are the loop; updates ARE the tracking. Anything
1303
+ // that is not an explicit update (absent/malformed action) is counted as
1304
+ // a create, so the original protection is unchanged for the loop case.
1305
+ if (call.name === 'plan_todo') {
1306
+ const action = (() => {
1307
+ try {
1308
+ const raw = call.arguments;
1309
+ const obj = typeof raw === 'string' ? JSON.parse(raw) : raw;
1310
+ return obj && typeof obj === 'object'
1311
+ ? String(obj.action ?? '')
1312
+ : undefined;
1313
+ }
1314
+ catch {
1315
+ return undefined;
1316
+ }
1317
+ })();
1318
+ const isUpdate = action === 'update';
1319
+ if (!isUpdate && planTodoCreates >= 1) {
1320
+ return {
1321
+ call,
1322
+ refuse: 'Error: a plan already exists for this turn — do NOT declare it again. Advance it instead: call plan_todo with action "update", the step id and its status, then keep doing the work.',
1323
+ };
1324
+ }
1325
+ if (isUpdate && planTodoUpdates >= PLAN_TODO_UPDATE_CAP) {
1326
+ return {
1327
+ call,
1328
+ refuse: `Error: plan_todo has been updated ${PLAN_TODO_UPDATE_CAP} times this turn — stop updating the plan and finish the remaining work.`,
1329
+ };
1330
+ }
1331
+ if (isUpdate)
1332
+ planTodoUpdates += 1;
1333
+ else
1334
+ planTodoCreates += 1;
1241
1335
  }
1242
1336
  if (tool?.category === 'pipeline' && priorSameTool >= 1) {
1243
1337
  // The guard this replaces tested the literal name `pipeline`, which is
@@ -1431,6 +1525,118 @@ async function runToolLoopInner(opts, progress) {
1431
1525
  toolSpan?.end({ ok: false, message: 'tool outcome was never reported' });
1432
1526
  }
1433
1527
  };
1528
+ // ── PROJECT PREREQUISITE PRE-FLIGHT (D2.2) ──────────────────────────
1529
+ // Before the FIRST build of the turn, check the project's markers so a
1530
+ // missing prerequisite (build.rs, a Cargo feature, an icon) is handed over
1531
+ // BEFORE a failed build — not after ten identical retries. Best-effort and
1532
+ // bounded: once per turn, only when a build is actually planned.
1533
+ if (!prerequisitesPreflighted) {
1534
+ const buildPlan = plans.find((p) => {
1535
+ if (p.refuse !== undefined)
1536
+ return false;
1537
+ if (p.call.name !== 'run_terminal' && p.call.name !== 'terminal')
1538
+ return false;
1539
+ try {
1540
+ const raw = p.call.arguments;
1541
+ const obj = typeof raw === 'string' ? JSON.parse(raw) : raw;
1542
+ const cmd = String(obj?.command ?? '');
1543
+ return isBuildCommand(cmd);
1544
+ }
1545
+ catch {
1546
+ return false;
1547
+ }
1548
+ });
1549
+ if (buildPlan) {
1550
+ prerequisitesPreflighted = true;
1551
+ try {
1552
+ const findings = checkProjectPrerequisites(createNodePrereqFs(opts.context?.cwd || process.cwd()));
1553
+ if (findings.length > 0) {
1554
+ deferredUserNudges.push(formatPreflightFindings(findings));
1555
+ deps.onEvent?.(` 🧱 pre-flight: ${findings.length} missing project prerequisite(s) — handed the exact fix before building.`);
1556
+ traceEvent({
1557
+ kind: 'gate',
1558
+ gate: 'prerequisite',
1559
+ summary: `pre-flight found ${findings.length} missing project prerequisite(s) before the build: ` +
1560
+ findings.map((f) => f.id).join(', '),
1561
+ });
1562
+ }
1563
+ }
1564
+ catch {
1565
+ // A pre-flight must never break the turn.
1566
+ }
1567
+ }
1568
+ }
1569
+ // ── PLAN GATE (E2) ──────────────────────────────────────────────────
1570
+ // A workspace-directing turn about to MUTATE without having declared a plan
1571
+ // is stopped ONCE: its first FILE mutation is refused and the model is told
1572
+ // to plan first (`plan_todo`). Terminal commands still trigger the nudge but
1573
+ // are not blocked (see PLAN_GATED_TOOL_NAMES) — a build or a test is a step
1574
+ // a plan is meant to reach, not a change to gate. Bounded by design: the
1575
+ // SECOND attempt runs even without a plan (see `planBlocked`), so this guides
1576
+ // the model into the plan → track → verify contract instead of walling the
1577
+ // turn off. A batch that DECLARES a plan itself is never blocked (the model
1578
+ // is already doing the right thing), and `requirePlan:false` leaves the gate
1579
+ // byte-identical to not existing.
1580
+ const declaresPlan = !opts.context?.planStore?.snapshot?.() && plans.some((p) => {
1581
+ if (p.call.name !== 'plan_todo')
1582
+ return false;
1583
+ try {
1584
+ const raw = p.call.arguments;
1585
+ const obj = typeof raw === 'string' ? JSON.parse(raw) : raw;
1586
+ const action = obj && typeof obj === 'object' ? String(obj.action ?? '') : '';
1587
+ return action !== 'update';
1588
+ }
1589
+ catch {
1590
+ // Malformed args still count as a create attempt — the plan_todo guard
1591
+ // above already handles a repeated create, and a first one is honored.
1592
+ return true;
1593
+ }
1594
+ });
1595
+ if (!planNudged &&
1596
+ !declaresPlan &&
1597
+ opts.requirePlan !== false &&
1598
+ schemas.length > 0 &&
1599
+ !opts.context?.planStore?.snapshot?.() &&
1600
+ plans.some((p) => p.refuse === undefined && MUTATING_TOOL_NAMES.has(p.call.name)) &&
1601
+ requestRequiresWorkspaceAction(requestText, authorization.authorized)) {
1602
+ planNudged = true;
1603
+ // Hard requirement, spent ONCE: refuse the first batch's FILE mutations
1604
+ // so nothing is written before a plan exists. It bumps the step budget by
1605
+ // one because it forces a genuine re-answer (the model must declare the
1606
+ // plan and retry), exactly like the zero-action nudge.
1607
+ const blocked = [];
1608
+ if (!planBlocked) {
1609
+ planBlocked = true;
1610
+ for (const p of plans) {
1611
+ if (p.refuse === undefined && PLAN_GATED_TOOL_NAMES.has(p.call.name)) {
1612
+ p.refuse =
1613
+ 'Error: declare a plan first via plan_todo — call it with 2–6 ordered steps (action "create"), then retry this change. The plan is how the turn is tracked and verified.';
1614
+ blocked.push(p.call.name);
1615
+ }
1616
+ }
1617
+ if (blocked.length > 0)
1618
+ stepLimit += 1;
1619
+ for (const name of blocked) {
1620
+ traceEvent({
1621
+ kind: 'refusal',
1622
+ gate: 'plan',
1623
+ tool: name,
1624
+ summary: `refused ${name}: no plan declared — the first mutation is blocked once so the model plans before it changes files`,
1625
+ });
1626
+ }
1627
+ }
1628
+ deps.onEvent?.(blocked.length > 0
1629
+ ? ' 🗺️ No plan declared — blocking the first change until the model plans.'
1630
+ : ' 🗺️ No plan declared — asking the model to plan before it changes files.');
1631
+ traceEvent({
1632
+ kind: 'gate',
1633
+ gate: 'plan',
1634
+ summary: blocked.length > 0
1635
+ ? `a workspace-directing turn was about to mutate without a plan — the first mutation batch (${blocked.join(', ')}) was refused once and the model asked to declare a plan first`
1636
+ : 'a workspace-directing turn was about to mutate without a plan — one bounded nudge to declare one first',
1637
+ });
1638
+ deferredUserNudges.push(planRequiredNudge(currentAsk(opts)));
1639
+ }
1434
1640
  const executed = new Array(plans.length).fill('');
1435
1641
  for (let i = 0; i < plans.length;) {
1436
1642
  // A run of consecutive read-only calls is ONE bounded fan-out; any
@@ -1581,6 +1787,54 @@ async function runToolLoopInner(opts, progress) {
1581
1787
  const parallelTip = parallel.note(call.name, !resultText.startsWith('Error:'));
1582
1788
  if (parallelTip)
1583
1789
  resultText = `${resultText}\n\n${parallelTip}`;
1790
+ // ── MISSING-PREREQUISITE TAKEOVER ──────────────────────────────────
1791
+ // A `command not found` (exit 127) for a KNOWN, installable system tool
1792
+ // is a STEP, not a wall — but a model can read it as an environment
1793
+ // limit and stop: "I cannot install system-level software on your host
1794
+ // machine — I am physically unable" (live: trace-1791127992452-qzgodi,
1795
+ // where a user who had explicitly granted terminal permission was handed
1796
+ // a manual `curl … | sh` step instead of the install being done). Inject
1797
+ // the exact install command and forbid the refusal, deterministically, so
1798
+ // the NEXT model step acts. Advisory text only — the loop never runs the
1799
+ // install itself; that stays the model's governed `run_terminal` call.
1800
+ if (!ranOk && (call.name === 'run_terminal' || call.name === 'terminal')) {
1801
+ const failedCommand = String(call.arguments?.command ?? '');
1802
+ // EVERY installable tool the failing line needs, not just the first —
1803
+ // a `cd app && cargo build && cmake .` turn should hand the model all
1804
+ // the prerequisites in one takeover instead of one per round-trip.
1805
+ const missing = installableToolsFromFailure(failedCommand, rawResult);
1806
+ if (missing.length > 0) {
1807
+ const installCommands = missing.map((t) => resolveInstallCommand(t));
1808
+ resultText = `${resultText}\n\n${toolTakeoverInstruction(missing, installCommands)}`;
1809
+ deps.onEvent?.(` 🛠️ ${missing.map((t) => `'${t}'`).join(', ')} missing and installable — handing the model the install command(s) to run.`);
1810
+ traceEvent({
1811
+ kind: 'gate',
1812
+ gate: 'tool-takeover',
1813
+ summary: `${missing.join(', ')} missing (command not found) and installable — the model is ` +
1814
+ 'told to install them itself instead of declaring itself unable',
1815
+ });
1816
+ }
1817
+ // ── PROJECT PREREQUISITE BACKSTOP (D2.3) ──────────────────────────
1818
+ // No missing BINARY, but this failure may be a missing project FILE/
1819
+ // FEATURE (the other half of the same wall): `src-tauri/build.rs`, a
1820
+ // Cargo `custom-protocol` feature, an icon — none of which is a tool to
1821
+ // install, so the takeover above never fired and the live macOS turn
1822
+ // re-ran `npx tauri build` ~10 times. Match the OUTPUT SIGNATURE, hand
1823
+ // the model the exact fix on the FIRST failure, and record it.
1824
+ if (missing.length === 0) {
1825
+ const rules = matchPrerequisiteSignatures(rawResult);
1826
+ if (rules.length > 0) {
1827
+ resultText = `${resultText}\n\n${prerequisiteTakeoverInstruction(rules)}`;
1828
+ deps.onEvent?.(` 🧱 build prerequisite signature matched (${rules.map((r) => r.id).join(', ')}) — handing the model the exact fix.`);
1829
+ traceEvent({
1830
+ kind: 'gate',
1831
+ gate: 'prerequisite',
1832
+ summary: `build failed on a missing PROJECT prerequisite (${rules.map((r) => r.id).join(', ')}) — ` +
1833
+ 'the model is given the exact fix instead of retrying the identical command',
1834
+ });
1835
+ }
1836
+ }
1837
+ }
1584
1838
  delivered[i] = resultText;
1585
1839
  // G18 — record what this call DID, in the model's own call order, with the
1586
1840
  // evidence a later reader needs: the args the gate saw, the first line of
@@ -1602,6 +1856,11 @@ async function runToolLoopInner(opts, progress) {
1602
1856
  });
1603
1857
  thread.push({ role: 'tool', toolCallId: call.id, content: resultText });
1604
1858
  }
1859
+ // Flush the planning-time nudges now that the assistant's tool_calls group is
1860
+ // complete (every call id has a `tool` result). This is the first point a
1861
+ // user/system message may safely follow the batch.
1862
+ for (const nudge of deferredUserNudges)
1863
+ thread.push({ role: 'user', content: nudge });
1605
1864
  // Track the generalized stall: a step that RAN tools but succeeded at none
1606
1865
  // of them extends the streak; any success clears it. A text-only step (no
1607
1866
  // tools) leaves it as it was — the model is thinking, not failing.
@@ -2298,6 +2557,22 @@ function zeroActionGateApplies(opts, progress, schemaCount, requestText, authori
2298
2557
  return false;
2299
2558
  return requestRequiresWorkspaceAction(requestText, authorized);
2300
2559
  }
2560
+ /**
2561
+ * The bounded PLAN-REQUIRED nudge (E2) — one advisory step telling the model to
2562
+ * declare a short plan before it mutates.
2563
+ *
2564
+ * Names the ask, states the contract (plan first, then work), and closes the
2565
+ * failure mode the nudge exists for: planning that REPLACES the work; the last
2566
+ * line forbids exactly that dormancy.
2567
+ */
2568
+ export function planRequiredNudge(ask) {
2569
+ const quoted = (ask || '').trim().replace(/\s+/g, ' ').slice(0, 300);
2570
+ return ('Before you change anything, declare a short PLAN for this turn — call `plan_todo` with 2–6 ordered steps' +
2571
+ ' (create the plan once, then advance it with action "update" as you go).' +
2572
+ (quoted ? ` The request: "${quoted}".` : '') +
2573
+ '\nA plan is how the user can follow along and how the turn is verified — keep it accurate, do not pad it,' +
2574
+ ' and do NOT let planning replace the work: declare it, then do the first step now.');
2575
+ }
2301
2576
  /**
2302
2577
  * The bounded ZERO-ACTION correction — the loop telling the model, in one step,
2303
2578
  * that a directed request has not been touched yet.