agent-nuvira 3.3.0 → 3.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (231) hide show
  1. package/.agents/skills/code-assessment/SKILL.md +10 -1
  2. package/.agents/skills/index.json +1 -1
  3. package/README.md +47 -9
  4. package/dist/agents/agents/runner.d.ts +9 -0
  5. package/dist/agents/agents/runner.d.ts.map +1 -1
  6. package/dist/agents/agents/runner.js +34 -12
  7. package/dist/agents/agents/runner.js.map +1 -1
  8. package/dist/agents/agents/writer-tool-calling.d.ts.map +1 -1
  9. package/dist/agents/agents/writer-tool-calling.js +7 -2
  10. package/dist/agents/agents/writer-tool-calling.js.map +1 -1
  11. package/dist/agents/agents/writer.d.ts.map +1 -1
  12. package/dist/agents/agents/writer.js +7 -1
  13. package/dist/agents/agents/writer.js.map +1 -1
  14. package/dist/agents/artifact-verification.d.ts +102 -0
  15. package/dist/agents/artifact-verification.d.ts.map +1 -0
  16. package/dist/agents/artifact-verification.js +216 -0
  17. package/dist/agents/artifact-verification.js.map +1 -0
  18. package/dist/agents/checkpoint-store.d.ts +73 -0
  19. package/dist/agents/checkpoint-store.d.ts.map +1 -1
  20. package/dist/agents/checkpoint-store.js +161 -0
  21. package/dist/agents/checkpoint-store.js.map +1 -1
  22. package/dist/agents/credential-store.d.ts +120 -0
  23. package/dist/agents/credential-store.d.ts.map +1 -1
  24. package/dist/agents/credential-store.js +358 -19
  25. package/dist/agents/credential-store.js.map +1 -1
  26. package/dist/agents/orchestrator.d.ts +10 -1
  27. package/dist/agents/orchestrator.d.ts.map +1 -1
  28. package/dist/agents/orchestrator.js +234 -75
  29. package/dist/agents/orchestrator.js.map +1 -1
  30. package/dist/agents/phase-engine.d.ts +61 -0
  31. package/dist/agents/phase-engine.d.ts.map +1 -1
  32. package/dist/agents/phase-engine.js +68 -2
  33. package/dist/agents/phase-engine.js.map +1 -1
  34. package/dist/agents/prompt-assembly.d.ts +11 -0
  35. package/dist/agents/prompt-assembly.d.ts.map +1 -1
  36. package/dist/agents/prompt-assembly.js +17 -0
  37. package/dist/agents/prompt-assembly.js.map +1 -1
  38. package/dist/agents/release-preflight.d.ts +121 -0
  39. package/dist/agents/release-preflight.d.ts.map +1 -0
  40. package/dist/agents/release-preflight.js +259 -0
  41. package/dist/agents/release-preflight.js.map +1 -0
  42. package/dist/agents/release-runner.d.ts +157 -0
  43. package/dist/agents/release-runner.d.ts.map +1 -0
  44. package/dist/agents/release-runner.js +719 -0
  45. package/dist/agents/release-runner.js.map +1 -0
  46. package/dist/agents/step-handoff.d.ts +190 -0
  47. package/dist/agents/step-handoff.d.ts.map +1 -0
  48. package/dist/agents/step-handoff.js +443 -0
  49. package/dist/agents/step-handoff.js.map +1 -0
  50. package/dist/cli/benchmark.d.ts.map +1 -1
  51. package/dist/cli/benchmark.js +11 -4
  52. package/dist/cli/benchmark.js.map +1 -1
  53. package/dist/cli/chat.d.ts.map +1 -1
  54. package/dist/cli/chat.js +25 -4
  55. package/dist/cli/chat.js.map +1 -1
  56. package/dist/cli/cli-program.d.ts.map +1 -1
  57. package/dist/cli/cli-program.js +9 -2
  58. package/dist/cli/cli-program.js.map +1 -1
  59. package/dist/cli/commands.d.ts +2 -1
  60. package/dist/cli/commands.d.ts.map +1 -1
  61. package/dist/cli/commands.js +3 -2
  62. package/dist/cli/commands.js.map +1 -1
  63. package/dist/cli/config.js +2 -2
  64. package/dist/cli/config.js.map +1 -1
  65. package/dist/cli/credentials.d.ts +28 -0
  66. package/dist/cli/credentials.d.ts.map +1 -0
  67. package/dist/cli/credentials.js +213 -0
  68. package/dist/cli/credentials.js.map +1 -0
  69. package/dist/cli/dashboard.d.ts +26 -0
  70. package/dist/cli/dashboard.d.ts.map +1 -1
  71. package/dist/cli/dashboard.js +123 -3
  72. package/dist/cli/dashboard.js.map +1 -1
  73. package/dist/cli/failover-runner.d.ts.map +1 -1
  74. package/dist/cli/failover-runner.js +9 -3
  75. package/dist/cli/failover-runner.js.map +1 -1
  76. package/dist/cli/intent.d.ts +1 -1
  77. package/dist/cli/intent.js +2 -2
  78. package/dist/cli/intent.js.map +1 -1
  79. package/dist/cli/loop-executor.d.ts +7 -0
  80. package/dist/cli/loop-executor.d.ts.map +1 -1
  81. package/dist/cli/loop-executor.js +75 -7
  82. package/dist/cli/loop-executor.js.map +1 -1
  83. package/dist/cli/nlu.d.ts.map +1 -1
  84. package/dist/cli/nlu.js +8 -2
  85. package/dist/cli/nlu.js.map +1 -1
  86. package/dist/cli/process-control.d.ts +15 -0
  87. package/dist/cli/process-control.d.ts.map +1 -1
  88. package/dist/cli/process-control.js +38 -14
  89. package/dist/cli/process-control.js.map +1 -1
  90. package/dist/cli/publish.d.ts +22 -0
  91. package/dist/cli/publish.d.ts.map +1 -1
  92. package/dist/cli/publish.js +109 -13
  93. package/dist/cli/publish.js.map +1 -1
  94. package/dist/config/live-credentials.d.ts +57 -0
  95. package/dist/config/live-credentials.d.ts.map +1 -0
  96. package/dist/config/live-credentials.js +126 -0
  97. package/dist/config/live-credentials.js.map +1 -0
  98. package/dist/config/paths.d.ts +18 -0
  99. package/dist/config/paths.d.ts.map +1 -1
  100. package/dist/config/paths.js +25 -0
  101. package/dist/config/paths.js.map +1 -1
  102. package/dist/federation/a2a-types.js +1 -1
  103. package/dist/federation/a2a-types.js.map +1 -1
  104. package/dist/forwarded.d.ts +2 -0
  105. package/dist/forwarded.d.ts.map +1 -0
  106. package/dist/forwarded.js +2 -0
  107. package/dist/forwarded.js.map +1 -0
  108. package/dist/inference/model-validator.d.ts +6 -1
  109. package/dist/inference/model-validator.d.ts.map +1 -1
  110. package/dist/inference/model-validator.js +7 -2
  111. package/dist/inference/model-validator.js.map +1 -1
  112. package/dist/inference/route-resolver.d.ts +116 -0
  113. package/dist/inference/route-resolver.d.ts.map +1 -0
  114. package/dist/inference/route-resolver.js +159 -0
  115. package/dist/inference/route-resolver.js.map +1 -0
  116. package/dist/inference/tool-call-utils.d.ts +44 -0
  117. package/dist/inference/tool-call-utils.d.ts.map +1 -1
  118. package/dist/inference/tool-call-utils.js +126 -4
  119. package/dist/inference/tool-call-utils.js.map +1 -1
  120. package/dist/learning/autonomy-policy.d.ts +24 -0
  121. package/dist/learning/autonomy-policy.d.ts.map +1 -1
  122. package/dist/learning/autonomy-policy.js +110 -23
  123. package/dist/learning/autonomy-policy.js.map +1 -1
  124. package/dist/learning/eval-framework.d.ts +24 -0
  125. package/dist/learning/eval-framework.d.ts.map +1 -1
  126. package/dist/learning/eval-framework.js +291 -0
  127. package/dist/learning/eval-framework.js.map +1 -1
  128. package/dist/learning/intent-envelope.d.ts +162 -0
  129. package/dist/learning/intent-envelope.d.ts.map +1 -0
  130. package/dist/learning/intent-envelope.js +235 -0
  131. package/dist/learning/intent-envelope.js.map +1 -0
  132. package/dist/learning/reasoning-trace.d.ts +2 -2
  133. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  134. package/dist/learning/reasoning-trace.js +25 -1
  135. package/dist/learning/reasoning-trace.js.map +1 -1
  136. package/dist/learning/resilient-call.d.ts.map +1 -1
  137. package/dist/learning/resilient-call.js +29 -17
  138. package/dist/learning/resilient-call.js.map +1 -1
  139. package/dist/learning/run-trace.d.ts +164 -0
  140. package/dist/learning/run-trace.d.ts.map +1 -0
  141. package/dist/learning/run-trace.js +342 -0
  142. package/dist/learning/run-trace.js.map +1 -0
  143. package/dist/learning/skill-store.d.ts.map +1 -1
  144. package/dist/learning/skill-store.js +35 -10
  145. package/dist/learning/skill-store.js.map +1 -1
  146. package/dist/learning/skill-types.d.ts +14 -0
  147. package/dist/learning/skill-types.d.ts.map +1 -1
  148. package/dist/learning/skill-types.js +19 -0
  149. package/dist/learning/skill-types.js.map +1 -1
  150. package/dist/mcp/catalog.js +2 -3
  151. package/dist/mcp/catalog.js.map +1 -1
  152. package/dist/mcp/manager.d.ts.map +1 -1
  153. package/dist/mcp/manager.js +5 -3
  154. package/dist/mcp/manager.js.map +1 -1
  155. package/dist/resources/command-manifest.json +10 -10
  156. package/dist/skills/bundled-skills.d.ts.map +1 -1
  157. package/dist/skills/bundled-skills.js +13 -2
  158. package/dist/skills/bundled-skills.js.map +1 -1
  159. package/dist/skills/secret-capture.d.ts.map +1 -1
  160. package/dist/skills/secret-capture.js +9 -4
  161. package/dist/skills/secret-capture.js.map +1 -1
  162. package/dist/tools/coding-tools.d.ts.map +1 -1
  163. package/dist/tools/coding-tools.js +33 -8
  164. package/dist/tools/coding-tools.js.map +1 -1
  165. package/dist/tools/credentials-tool.d.ts +24 -0
  166. package/dist/tools/credentials-tool.d.ts.map +1 -0
  167. package/dist/tools/credentials-tool.js +125 -0
  168. package/dist/tools/credentials-tool.js.map +1 -0
  169. package/dist/tools/edit-verification.d.ts +78 -0
  170. package/dist/tools/edit-verification.d.ts.map +1 -1
  171. package/dist/tools/edit-verification.js +213 -16
  172. package/dist/tools/edit-verification.js.map +1 -1
  173. package/dist/tools/git-tool.d.ts +25 -7
  174. package/dist/tools/git-tool.d.ts.map +1 -1
  175. package/dist/tools/git-tool.js +155 -15
  176. package/dist/tools/git-tool.js.map +1 -1
  177. package/dist/tools/loop-project-context.d.ts.map +1 -1
  178. package/dist/tools/loop-project-context.js +16 -0
  179. package/dist/tools/loop-project-context.js.map +1 -1
  180. package/dist/tools/loop-route-feed.d.ts +57 -0
  181. package/dist/tools/loop-route-feed.d.ts.map +1 -0
  182. package/dist/tools/loop-route-feed.js +101 -0
  183. package/dist/tools/loop-route-feed.js.map +1 -0
  184. package/dist/tools/pipeline-tool.d.ts.map +1 -1
  185. package/dist/tools/pipeline-tool.js +8 -2
  186. package/dist/tools/pipeline-tool.js.map +1 -1
  187. package/dist/tools/publish-tool.d.ts.map +1 -1
  188. package/dist/tools/publish-tool.js +76 -10
  189. package/dist/tools/publish-tool.js.map +1 -1
  190. package/dist/tools/registry.d.ts +52 -3
  191. package/dist/tools/registry.d.ts.map +1 -1
  192. package/dist/tools/registry.js +88 -11
  193. package/dist/tools/registry.js.map +1 -1
  194. package/dist/tools/run-cli.d.ts.map +1 -1
  195. package/dist/tools/run-cli.js +15 -2
  196. package/dist/tools/run-cli.js.map +1 -1
  197. package/dist/tools/run-terminal.d.ts +9 -0
  198. package/dist/tools/run-terminal.d.ts.map +1 -1
  199. package/dist/tools/run-terminal.js +103 -6
  200. package/dist/tools/run-terminal.js.map +1 -1
  201. package/dist/tools/skill-tool.d.ts.map +1 -1
  202. package/dist/tools/skill-tool.js +5 -0
  203. package/dist/tools/skill-tool.js.map +1 -1
  204. package/dist/tools/tool-loop.d.ts +40 -6
  205. package/dist/tools/tool-loop.d.ts.map +1 -1
  206. package/dist/tools/tool-loop.js +409 -14
  207. package/dist/tools/tool-loop.js.map +1 -1
  208. package/dist/tools/toolsets.js +2 -2
  209. package/dist/tools/toolsets.js.map +1 -1
  210. package/dist/web-dashboard/chat-console.d.ts +5 -0
  211. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  212. package/dist/web-dashboard/chat-console.js +32 -3
  213. package/dist/web-dashboard/chat-console.js.map +1 -1
  214. package/dist/web-dashboard/server.d.ts +46 -0
  215. package/dist/web-dashboard/server.d.ts.map +1 -1
  216. package/dist/web-dashboard/server.js +115 -1
  217. package/dist/web-dashboard/server.js.map +1 -1
  218. package/dist/web-dashboard/src/admin-auth.d.ts +49 -2
  219. package/dist/web-dashboard/src/admin-auth.d.ts.map +1 -1
  220. package/dist/web-dashboard/src/admin-auth.js +64 -3
  221. package/dist/web-dashboard/src/admin-auth.js.map +1 -1
  222. package/dist/web-dashboard/src/types.d.ts +41 -0
  223. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  224. package/dist/workflow/registry.js +1 -1
  225. package/dist/workflow/registry.js.map +1 -1
  226. package/package.json +9 -5
  227. package/src/web-dashboard/public/assets/index-CxDj7p6i.js +207 -0
  228. package/src/web-dashboard/public/assets/index-CxDj7p6i.js.map +1 -0
  229. package/src/web-dashboard/public/index.html +1 -1
  230. package/src/web-dashboard/public/assets/index-kCUkORm7.js +0 -207
  231. package/src/web-dashboard/public/assets/index-kCUkORm7.js.map +0 -1
@@ -17,6 +17,7 @@
17
17
  * Called by the `agent-nuvira execute` CLI command.
18
18
  */
19
19
  import { existsSync, readFileSync, writeFileSync, mkdirSync, readdirSync, statSync } from 'node:fs';
20
+ import { verifyArtifacts } from './artifact-verification.js';
20
21
  import { dirname, isAbsolute, join, relative, resolve } from 'node:path';
21
22
  import { spawnSync } from 'node:child_process';
22
23
  import { pushDAGUpdate, updateDAGNode, resetDAG } from '../observability/dag-bridge.js';
@@ -27,7 +28,7 @@ import { showModelPicker } from '../cli/model-picker.js';
27
28
  import { shouldPromptWeakModel, promptWeakModelChoice } from '../cli/weak-model-prompt.js';
28
29
  import { logger } from '../utils/logger.js';
29
30
  import { ContextVault } from './context-vault.js';
30
- import { saveCheckpoint, loadCheckpoint, checkpointIdFor } from './checkpoint-store.js';
31
+ import { saveCheckpoint, loadCheckpoint, checkpointIdFor, findRelatedCheckpointFor, reconcileTaskPlan, planHasPendingWork, } from './checkpoint-store.js';
31
32
  import { buildLongFormPlan, isContinuationAsk, } from './long-form-plan.js';
32
33
  import { buildCompositePlan } from './composite-plan.js';
33
34
  import { assembleDocument, countWords, findInProgressJob, formatProgress, jobProgress, recordSectionOutcome, } from '../learning/long-form.js';
@@ -69,12 +70,13 @@ import { recordActionFailure } from '../learning/failure-bookkeeping.js';
69
70
  import { sweepTransientFailures, sessionRevivalStore } from '../learning/provider-revival.js';
70
71
  import { resolveContextBudget, resolveContextFileBudget, resolveMaxOutputTokens } from '../learning/context-budget.js';
71
72
  import { clampMaxTokens, learnMaxTokensLimitFromError } from '../learning/provider-limits.js';
72
- import { resolveWorkingModel } from '../inference/model-validator.js';
73
+ import { resolveRoute } from '../inference/route-resolver.js';
73
74
  import { refreshModelRegistry } from '../inference/model-probe.js';
74
75
  import { recordRoutingDecision } from '../learning/routing-history.js';
75
76
  import { getQuotaLedger } from '../learning/quota-ledger.js';
76
77
  import { withTraceCapture, beginTrace, endTrace } from '../learning/reasoning-trace.js';
77
78
  import { recordWorkingState } from '../learning/working-state.js';
79
+ import { clearStepHandoff, deliverablesNamedIn, recordStepHandoff, stepKeyFor } from './step-handoff.js';
78
80
  import { createResilientCallLLM } from '../learning/resilient-call.js';
79
81
  import { createReviewFromResult } from '../team/review.js';
80
82
  import { indexFiles, retrieve, recordRetrievalStats, retrievalOptionsFromConfig, estimateTokens as retrievalEstimateTokens } from '../learning/retrieval.js';
@@ -383,33 +385,88 @@ export class Orchestrator {
383
385
  const checkpointEnabled = options.checkpoint === true ||
384
386
  !!options.resumeCheckpointId ||
385
387
  options.resumeRequested === true;
386
- // LOAD only when the user explicitly asked to RESUME (bare --resume or an
387
- // explicit id). Plain `--checkpoint` must NEVER silently resume a stale
388
- // checkpoint from a previous run of the same goal — that would re-enter a
389
- // completed plan and skip every task.
388
+ // ── The work ledger is LOADED AND RECONCILED ON EVERY RUN ────────────────
389
+ // v1.62.4 loaded it only on an explicit `--resume`, and that gate was
390
+ // justified by a real hazard: silently re-entering a COMPLETED plan would
391
+ // skip every task. The hazard is real, but the gate treated the checkpoint's
392
+ // own `completed` flags as proof — and the live NVDA-addon checkpoint said
393
+ // `5/5 completed` while three of those steps had produced nothing, so the
394
+ // one run that should have resumed was the one run forbidden from it. The
395
+ // same plan was then re-planned 18 times.
396
+ //
397
+ // The fix is to make the flags verifiable rather than to keep ignoring them:
398
+ // `reconcileTaskPlan` demotes a "completed" step whose DECLARED files are not
399
+ // on disk, and only then is the checkpoint consulted without being asked.
400
+ //
401
+ // * explicit --resume → load whatever is there, reconciled.
402
+ // * same goal, no --resume → load the auto-id checkpoint, reconciled;
403
+ // continue only if work is genuinely left.
404
+ // * REWORDED goal → the auto-id misses (it hashes goal + cwd),
405
+ // so fall back to the newest checkpoint for
406
+ // this PROJECT. This is what makes "do that
407
+ // addon thing again" resume instead of
408
+ // re-planning from zero.
409
+ // * everything verified done → start fresh (the old guard, now earned).
390
410
  const resumeWanted = options.resumeRequested === true || !!options.resumeCheckpointId;
391
411
  let resumed = false;
392
412
  let vault;
393
- if (resumeWanted) {
394
- const saved = loadCheckpoint(resumeId);
395
- if (saved) {
396
- vault = ContextVault.fromSnapshot(saved.context);
397
- resumed = true;
398
- const done = saved.context.taskPlan.filter((s) => s.status === 'completed').length;
399
- if (options.verbose) {
400
- logger.info(` ♻️ Resumed from checkpoint '${resumeId}' — ${done}/${saved.context.taskPlan.length} steps already complete`);
413
+ {
414
+ let saved = loadCheckpoint(resumeId);
415
+ let source = 'goal';
416
+ if (!saved && !resumeWanted) {
417
+ // A reworded ask hashes to a different id — look for the same ask
418
+ // recorded differently. The goal test is required: taking the newest
419
+ // checkpoint in the directory would hand this run the plan of an
420
+ // unrelated run that merely shared a folder.
421
+ try {
422
+ saved = findRelatedCheckpointFor(process.cwd(), goal, { excludeId: resumeId });
423
+ if (saved)
424
+ source = 'project';
425
+ }
426
+ catch {
427
+ saved = null;
401
428
  }
402
429
  }
403
- else {
404
- // Resume explicitly requested but no checkpoint found — warn (a
405
- // reworded goal silently misses the auto id) and start fresh with
406
- // checkpointing on, so a later crash can still be resumed.
407
- logger.warn(` ⚠️ No checkpoint found for '${resumeId}' — starting a fresh pipeline (run with --checkpoint to save one)`);
430
+ if (!saved) {
431
+ if (resumeWanted) {
432
+ // Resume explicitly requested but no checkpoint found — warn (a
433
+ // reworded goal silently misses the auto id) and start fresh with
434
+ // checkpointing on, so a later crash can still be resumed.
435
+ logger.warn(` ⚠️ No checkpoint found for '${resumeId}' — starting a fresh pipeline (run with --checkpoint to save one)`);
436
+ }
408
437
  vault = new ContextVault(goal, process.cwd());
409
438
  }
410
- }
411
- else {
412
- vault = new ContextVault(goal, process.cwd());
439
+ else {
440
+ const firstStep = saved.context.taskPlan.length;
441
+ const { context: reconciledContext, demoted } = reconcileTaskPlan(saved.context, process.cwd());
442
+ const workLeft = demoted.length > 0 || planHasPendingWork(reconciledContext);
443
+ // An explicit resume always continues. An automatic one continues only
444
+ // when reconciliation found real work — a plan that verifies as finished
445
+ // must not be re-entered (that is the hazard the old gate guarded).
446
+ if (resumeWanted || workLeft) {
447
+ vault = ContextVault.fromSnapshot(reconciledContext);
448
+ resumed = true;
449
+ const done = reconciledContext.taskPlan.filter((s) => s.status === 'completed').length;
450
+ if (demoted.length > 0) {
451
+ // Named, never silent: the checkpoint claimed these were done and
452
+ // the filesystem disagrees. That gap IS the bug being fixed.
453
+ logger.warn(` ♻️ Re-opened ${demoted.length} of ${firstStep} step(s) from checkpoint '${saved.id}' — "completed" but the deliverable is not on disk:`);
454
+ for (const d of demoted)
455
+ logger.warn(` ⛔ ${d.id}: ${d.reason}`);
456
+ }
457
+ if (options.verbose || demoted.length > 0) {
458
+ logger.info(` ♻️ ${source === 'project' ? 'Continuing this project\'s' : 'Resumed from'} checkpoint '${saved.id}' — ${done}/${reconciledContext.taskPlan.length} steps verified complete`);
459
+ }
460
+ }
461
+ else {
462
+ // Every declared artifact is on disk — the plan is genuinely finished,
463
+ // so a fresh run starts fresh rather than skipping work it should do.
464
+ if (options.verbose) {
465
+ logger.info(` ℹ️ Checkpoint '${saved.id}' verifies as complete — starting a fresh plan`);
466
+ }
467
+ vault = new ContextVault(goal, process.cwd());
468
+ }
469
+ }
413
470
  }
414
471
  // ── Transparency channel ─────────────────────────────────────────────
415
472
  // Agents call context.onAgentUpdate() (via Agent.report()) to stream
@@ -1181,7 +1238,28 @@ export class Orchestrator {
1181
1238
  // the COMPLETED state (including the final batch). Without this, the last
1182
1239
  // saved checkpoint would show the final step still 'pending', and a
1183
1240
  // --resume after a successful run would re-execute it.
1184
- if (checkpointEnabled) {
1241
+ //
1242
+ // ALSO saved when the run ends with work UNFINISHED, whether or not the user
1243
+ // opted into checkpoints. A failed step, or a "completed" step whose declared
1244
+ // deliverable is not on disk (see `reconcileTaskPlan`), means the ledger is
1245
+ // the only thing standing between this run and the next one re-deriving the
1246
+ // whole plan — which is exactly what happened 18 times on the live NVDA
1247
+ // add-on ask. Bounded on purpose: a run that finished cleanly and opted out
1248
+ // writes nothing, so the store does not grow on every invocation.
1249
+ let persistLedger = checkpointEnabled;
1250
+ if (!persistLedger) {
1251
+ try {
1252
+ const { demoted } = reconcileTaskPlan(vault.context, process.cwd());
1253
+ persistLedger = vault.hasFailedTasks || demoted.length > 0;
1254
+ if (persistLedger && options.verbose) {
1255
+ logger.info(` 💾 Work is unfinished (${vault.hasFailedTasks ? 'a step failed' : `${demoted.length} step(s) missing their deliverable`}) — saving the ledger for the next run`);
1256
+ }
1257
+ }
1258
+ catch {
1259
+ // Best-effort — the ledger must never break result delivery.
1260
+ }
1261
+ }
1262
+ if (persistLedger) {
1185
1263
  try {
1186
1264
  const cid = saveCheckpoint(vault.context, resumeId);
1187
1265
  if (cid && options.verbose) {
@@ -1637,19 +1715,33 @@ export class Orchestrator {
1637
1715
  // Resolve it to the provider's configured model (or best available) so
1638
1716
  // planner/memory/rate-limit-switch calls never crash with "no auto model".
1639
1717
  let requestedModel = options.model || inferenceOptions?.model || config.model;
1640
- // CRITICAL FIX: When no model is specified (undefined) or the sentinel 'default'
1641
- // is used, resolve via the provider's live model list. This ensures the pipeline
1642
- // never sends a literal 'default' or undefined to a provider API, which would 404.
1643
- if (!requestedModel || requestedModel === 'default') {
1644
- try {
1645
- const resolved = await resolveWorkingModel(provider, providerType, requestedModel);
1646
- requestedModel = resolved;
1647
- }
1648
- catch {
1649
- // Best-effort — fall through to the original value if resolution fails
1650
- }
1651
- }
1652
- const servedModel = isAutoModel(requestedModel) ? (config.model || requestedModel) : requestedModel;
1718
+ // 'auto' is a DIRECTIVE, not a model id: resolve it to the provider's
1719
+ // configured model BEFORE validating, so the validator is never asked to
1720
+ // repair a sentinel and the 'auto is not available on X' warning is not
1721
+ // printed for a value the user never chose as a model.
1722
+ if (isAutoModel(requestedModel)) {
1723
+ requestedModel = config.model || undefined;
1724
+ }
1725
+ // ALWAYS validate the pair — the model against THIS provider instance,
1726
+ // which is the one that will serve the call.
1727
+ //
1728
+ // This used to run only when the model was empty or the literal 'default',
1729
+ // which skipped the validator in exactly the case it was written for: a
1730
+ // real, non-empty model id that does not belong to this provider (issue
1731
+ // #10). A stale or foreign pin then went straight to the API and 404'd —
1732
+ // `Groq API error (404): The model 'gemini-3.1-flash-lite' does not exist`
1733
+ // — with the repair machinery sitting right there, unused.
1734
+ //
1735
+ // resolveRoute also owns the reporting: a substitution is printed and
1736
+ // recorded rather than made silently, and NUVIRA_STRICT_MODEL=1 refuses to
1737
+ // substitute at all (issue #11).
1738
+ const route = await resolveRoute({
1739
+ providerType,
1740
+ provider,
1741
+ model: requestedModel,
1742
+ source: 'orchestrator',
1743
+ });
1744
+ const servedModel = route.model;
1653
1745
  const mergedOptions = {
1654
1746
  ...inferenceOptions,
1655
1747
  model: servedModel,
@@ -2457,26 +2549,61 @@ export class Orchestrator {
2457
2549
  else if (firstFailed) {
2458
2550
  stats.recoveredFailures += 1;
2459
2551
  }
2460
- // v1.62.4 — Deliverable verification: a writer step must actually
2461
- // produce its declared expectedFiles. Catches the live NVDA-addon bug
2462
- // where step "Create manifest.ini" reported success while writing
2463
- // globalPlugins/hello_anuj.py instead — a step that claims success
2464
- // without its deliverable is a FAILURE, so the pipeline repairs it
2465
- // instead of silently building on the wrong file. Checked BEFORE the
2466
- // result is pushed so agentResults/status reflect the corrected outcome.
2467
- if (effectiveAgentType === 'writer' && result.success && task.expectedFiles && task.expectedFiles.length > 0) {
2468
- const proposedPaths = vault.context.fileChanges
2469
- .filter((c) => c.status === 'created' || c.status === 'modified')
2470
- .map((c) => normalizeSlash(c.path));
2471
- const missing = task.expectedFiles.filter((f) => {
2472
- const norm = normalizeSlash(f);
2473
- // On disk (applied) OR proposed in this step's changes — either
2474
- // satisfies the deliverable.
2475
- return !proposedPaths.includes(norm) &&
2476
- !existsSync(isAbsolute(norm) ? norm : resolve(process.cwd(), norm));
2477
- });
2478
- if (missing.length > 0) {
2479
- const msg = `Deliverable mismatch — step claimed success but did not produce expected file(s): ${missing.join(', ')}`;
2552
+ // ── The step's OWN writes reach disk BEFORE anything judges them ───────
2553
+ // Ordering, and the reason it is load-bearing: the deliverable check just
2554
+ // below asks whether the declared artifacts EXIST on disk. A writer's
2555
+ // output only reaches disk in `applyFileChanges`, which used to run at the
2556
+ // END of this block — after the check. So the check asked about a file the
2557
+ // pipeline had not written yet, every creating writer step was reported as
2558
+ // "step claimed success but the artifact is not there", and because the
2559
+ // later apply is gated on `result.success` (now false) the write was then
2560
+ // SKIPPED as well: the step failed itself AND lost its work. Both real
2561
+ // long-form E2E tests caught it as "only unit 1 exists on disk".
2562
+ if (result.success && !options.dryRun && (effectiveAgentType === 'writer' || effectiveAgentType === 'debugger')) {
2563
+ const applied = this.applyFileChanges(vault);
2564
+ if (applied > 0 && options.verbose) {
2565
+ logger.info(` 💾 Applied ${applied} file change${applied !== 1 ? 's' : ''} to disk` +
2566
+ (effectiveAgentType === 'debugger' ? ' (debug fix)' : ''));
2567
+ }
2568
+ }
2569
+ // Deliverable verification: a step that declares files must produce them
2570
+ // ON DISK. Two holes in the v1.62.4 version of this guard let the live
2571
+ // NVDA-addon failure through, and both are closed here.
2572
+ //
2573
+ // 1. It accepted the agent's own REPORT of what it wrote. That run's
2574
+ // `fileChanges` claimed .../kuttaaddon/installTasks.py was `created`
2575
+ // and the file did not exist — but the claim satisfied the check, so
2576
+ // the disk test never ran. A claim is not an artifact: only the
2577
+ // filesystem counts now, and a path that was reported as written while
2578
+ // being absent is named explicitly, because that discrepancy IS the
2579
+ // bug rather than a side note.
2580
+ // 2. It only ran for `writer` steps. The step that fabricated the empty
2581
+ // package was a `runner` (`zip -r kuttaaddon.nvda-addon …`), so the one
2582
+ // step whose whole job was to produce the deliverable was structurally
2583
+ // exempt from the check. Any step may declare expectedFiles, so any
2584
+ // step is verified.
2585
+ //
2586
+ // The test is also no longer existence-only: an empty file is not a
2587
+ // deliverable, and a zip holding zero entries is not a package (see
2588
+ // artifact-verification.ts for how an empty archive is detected). Checked
2589
+ // BEFORE the result is recorded so agentResults, the task status and the
2590
+ // checkpoint all reflect the corrected outcome.
2591
+ if (result.success && task.expectedFiles && task.expectedFiles.length > 0) {
2592
+ const root = vault.context.workingDirectory || process.cwd();
2593
+ const check = verifyArtifacts(task.expectedFiles, root);
2594
+ if (!check.ok) {
2595
+ const reported = new Set(vault.context.fileChanges
2596
+ .filter((c) => c.status === 'created' || c.status === 'modified')
2597
+ .map((c) => normalizeSlash(c.path)));
2598
+ const claimedButAbsent = check.missing.filter((f) => reported.has(normalizeSlash(f)));
2599
+ const msg = [
2600
+ `Deliverable mismatch — step claimed success but the artifact is not there: ${check.reason}`,
2601
+ claimedButAbsent.length
2602
+ ? `(reported as written but absent from disk: ${claimedButAbsent.join(', ')})`
2603
+ : '',
2604
+ ]
2605
+ .filter(Boolean)
2606
+ .join(' ');
2480
2607
  logger.warn(` ⚠️ ${msg}`);
2481
2608
  result = {
2482
2609
  success: false,
@@ -2485,6 +2612,42 @@ export class Orchestrator {
2485
2612
  };
2486
2613
  }
2487
2614
  }
2615
+ // ── DURABLE HAND-OFF: this step's outcome survives the run ─────────────
2616
+ // A failed step writes down what it must produce and what is already on
2617
+ // disk; a succeeded step CLEARS its own entry. That pairing is the whole
2618
+ // mechanism — the live NVDA-addon run had no such record, so the next
2619
+ // attempt inherited nothing and re-derived the same plan (18 times), and
2620
+ // a step that finally succeeded would have stayed on the outstanding list
2621
+ // forever without the clear.
2622
+ //
2623
+ // Keyed on the step's DECLARED artifacts when it has them, so the same
2624
+ // deliverable asked for in different words maps to one hand-off. Root is
2625
+ // the vault's own working directory — the same root the deliverable check
2626
+ // above used, so the two can never disagree about where "on disk" means.
2627
+ try {
2628
+ const handoffRoot = vault.context.workingDirectory || process.cwd();
2629
+ const declared = (task.expectedFiles ?? []).filter((f) => f && f.trim());
2630
+ const runGoal = vault.context.goal;
2631
+ if (result.success) {
2632
+ if (declared.length > 0)
2633
+ clearStepHandoff(handoffRoot, stepKeyFor({ declared }));
2634
+ }
2635
+ else {
2636
+ const named = declared.length > 0 ? declared : deliverablesNamedIn(`${runGoal} ${task.description}`);
2637
+ recordStepHandoff({
2638
+ projectPath: handoffRoot,
2639
+ goal: runGoal,
2640
+ stepDescription: task.description,
2641
+ declared: named,
2642
+ route: effectiveAgentType,
2643
+ kind: result.error && /refus|denied|outside the workspace/i.test(result.error) ? 'refused' : 'failed',
2644
+ reason: result.error || result.summary,
2645
+ });
2646
+ }
2647
+ }
2648
+ catch {
2649
+ // Best-effort — the hand-off ledger must never break the pipeline.
2650
+ }
2488
2651
  vault.updateTaskStatus(task.id, result.success ? 'completed' : 'failed', result.summary);
2489
2652
  await tryUpdateDAGNode(task.id, {
2490
2653
  status: result.success ? 'completed' : 'failed',
@@ -2531,25 +2694,10 @@ export class Orchestrator {
2531
2694
  vault.setMeta('sandboxPath', testResult.sandboxPath);
2532
2695
  }
2533
2696
  }
2534
- // After debugger step: write debugger's fixes to disk immediately
2535
- // The DebuggerAgent's syncChangesToContext() updates context.fileChanges
2536
- // with LLM-generated fixes. If a runner step follows the debugger, those
2537
- // fixes must be on disk before the runner executes.
2538
- if (effectiveAgentType === 'debugger' && result.success && !options.dryRun) {
2539
- const applied = this.applyFileChanges(vault);
2540
- if (applied > 0 && options.verbose) {
2541
- logger.info(` 💾 Applied ${applied} debug fix(es) to disk`);
2542
- }
2543
- }
2544
- // After writer step: write files to disk immediately and sync into artifacts
2545
- // IMPORTANT: files MUST be on disk before the RunnerAgent tries to execute them
2697
+ // ── Writer artifacts sync (writer/debugger writes were applied ABOVE,
2698
+ // before the deliverable check, so a runner following this step still
2699
+ // finds them on disk) ─────────────────────────────────────────────────
2546
2700
  if (effectiveAgentType === 'writer' && result.success) {
2547
- if (!options.dryRun) {
2548
- const applied = this.applyFileChanges(vault);
2549
- if (applied > 0 && options.verbose) {
2550
- logger.info(` 💾 Applied ${applied} file change${applied !== 1 ? 's' : ''} to disk`);
2551
- }
2552
- }
2553
2701
  const newArtifacts = vault.context.fileChanges
2554
2702
  .filter((c) => c.status === 'created' || c.status === 'modified')
2555
2703
  .filter((c) => c.newContent)
@@ -3108,7 +3256,18 @@ export class Orchestrator {
3108
3256
  try {
3109
3257
  const { config } = this.configManager.getProviderConfig(decision.provider);
3110
3258
  const adapter = ProviderFactory.createProvider(decision.provider, config);
3111
- workingModel = await resolveWorkingModel(adapter, decision.provider, decision.model);
3259
+ // resolveRoute, not resolveWorkingModel: identical repair policy, but
3260
+ // the pair is validated against the adapter that will serve the call
3261
+ // and a substitution is printed + recorded instead of silent.
3262
+ const route = await resolveRoute({
3263
+ providerType: decision.provider,
3264
+ provider: adapter,
3265
+ model: decision.model,
3266
+ source: 'orchestrator',
3267
+ agentType: task.agentType,
3268
+ task: task.description,
3269
+ });
3270
+ workingModel = route.model;
3112
3271
  }
3113
3272
  catch {
3114
3273
  workingModel = decision.model;