agent-nuvira 3.1.3 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/dist/agents/agents/reasoner.d.ts +8 -0
  2. package/dist/agents/agents/reasoner.d.ts.map +1 -1
  3. package/dist/agents/agents/reasoner.js +53 -2
  4. package/dist/agents/agents/reasoner.js.map +1 -1
  5. package/dist/agents/agents/writer.d.ts +24 -0
  6. package/dist/agents/agents/writer.d.ts.map +1 -1
  7. package/dist/agents/agents/writer.js +194 -0
  8. package/dist/agents/agents/writer.js.map +1 -1
  9. package/dist/agents/composite-plan.d.ts +143 -0
  10. package/dist/agents/composite-plan.d.ts.map +1 -0
  11. package/dist/agents/composite-plan.js +399 -0
  12. package/dist/agents/composite-plan.js.map +1 -0
  13. package/dist/agents/long-form-plan.d.ts +156 -0
  14. package/dist/agents/long-form-plan.d.ts.map +1 -0
  15. package/dist/agents/long-form-plan.js +274 -0
  16. package/dist/agents/long-form-plan.js.map +1 -0
  17. package/dist/agents/orchestrator.d.ts +107 -0
  18. package/dist/agents/orchestrator.d.ts.map +1 -1
  19. package/dist/agents/orchestrator.js +501 -35
  20. package/dist/agents/orchestrator.js.map +1 -1
  21. package/dist/agents/prompt-assembly.d.ts +8 -0
  22. package/dist/agents/prompt-assembly.d.ts.map +1 -1
  23. package/dist/agents/prompt-assembly.js +17 -0
  24. package/dist/agents/prompt-assembly.js.map +1 -1
  25. package/dist/cli/chat.d.ts +27 -0
  26. package/dist/cli/chat.d.ts.map +1 -1
  27. package/dist/cli/chat.js +116 -6
  28. package/dist/cli/chat.js.map +1 -1
  29. package/dist/cli/execute.d.ts +12 -0
  30. package/dist/cli/execute.d.ts.map +1 -1
  31. package/dist/cli/execute.js +105 -2
  32. package/dist/cli/execute.js.map +1 -1
  33. package/dist/cli/models.d.ts.map +1 -1
  34. package/dist/cli/models.js +10 -0
  35. package/dist/cli/models.js.map +1 -1
  36. package/dist/cli/trace.d.ts.map +1 -1
  37. package/dist/cli/trace.js +27 -0
  38. package/dist/cli/trace.js.map +1 -1
  39. package/dist/gateway/registry.d.ts +31 -0
  40. package/dist/gateway/registry.d.ts.map +1 -1
  41. package/dist/gateway/registry.js +143 -13
  42. package/dist/gateway/registry.js.map +1 -1
  43. package/dist/inference/model-entitlement.d.ts +46 -0
  44. package/dist/inference/model-entitlement.d.ts.map +1 -0
  45. package/dist/inference/model-entitlement.js +98 -0
  46. package/dist/inference/model-entitlement.js.map +1 -0
  47. package/dist/learning/autonomy-policy.d.ts +334 -0
  48. package/dist/learning/autonomy-policy.d.ts.map +1 -0
  49. package/dist/learning/autonomy-policy.js +500 -0
  50. package/dist/learning/autonomy-policy.js.map +1 -0
  51. package/dist/learning/credential-fingerprint.d.ts +58 -0
  52. package/dist/learning/credential-fingerprint.d.ts.map +1 -0
  53. package/dist/learning/credential-fingerprint.js +126 -0
  54. package/dist/learning/credential-fingerprint.js.map +1 -0
  55. package/dist/learning/deliverable-class.d.ts +130 -0
  56. package/dist/learning/deliverable-class.d.ts.map +1 -0
  57. package/dist/learning/deliverable-class.js +432 -0
  58. package/dist/learning/deliverable-class.js.map +1 -0
  59. package/dist/learning/long-form.d.ts +252 -0
  60. package/dist/learning/long-form.d.ts.map +1 -0
  61. package/dist/learning/long-form.js +510 -0
  62. package/dist/learning/long-form.js.map +1 -0
  63. package/dist/learning/model-first-router.d.ts +17 -0
  64. package/dist/learning/model-first-router.d.ts.map +1 -1
  65. package/dist/learning/model-first-router.js +27 -0
  66. package/dist/learning/model-first-router.js.map +1 -1
  67. package/dist/learning/model-registry.d.ts +66 -4
  68. package/dist/learning/model-registry.d.ts.map +1 -1
  69. package/dist/learning/model-registry.js +67 -6
  70. package/dist/learning/model-registry.js.map +1 -1
  71. package/dist/learning/model-warmup.d.ts +97 -2
  72. package/dist/learning/model-warmup.d.ts.map +1 -1
  73. package/dist/learning/model-warmup.js +165 -60
  74. package/dist/learning/model-warmup.js.map +1 -1
  75. package/dist/learning/prompt-layers.d.ts +61 -0
  76. package/dist/learning/prompt-layers.d.ts.map +1 -0
  77. package/dist/learning/prompt-layers.js +140 -0
  78. package/dist/learning/prompt-layers.js.map +1 -0
  79. package/dist/learning/provider-limits.d.ts +66 -0
  80. package/dist/learning/provider-limits.d.ts.map +1 -0
  81. package/dist/learning/provider-limits.js +184 -0
  82. package/dist/learning/provider-limits.js.map +1 -0
  83. package/dist/learning/reasoning-trace.d.ts +36 -1
  84. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  85. package/dist/learning/reasoning-trace.js +36 -2
  86. package/dist/learning/reasoning-trace.js.map +1 -1
  87. package/dist/learning/resilient-call.d.ts +36 -1
  88. package/dist/learning/resilient-call.d.ts.map +1 -1
  89. package/dist/learning/resilient-call.js +80 -5
  90. package/dist/learning/resilient-call.js.map +1 -1
  91. package/dist/learning/unattended-job.d.ts +293 -0
  92. package/dist/learning/unattended-job.d.ts.map +1 -0
  93. package/dist/learning/unattended-job.js +544 -0
  94. package/dist/learning/unattended-job.js.map +1 -0
  95. package/dist/learning/unattended-progress.d.ts +95 -0
  96. package/dist/learning/unattended-progress.d.ts.map +1 -0
  97. package/dist/learning/unattended-progress.js +147 -0
  98. package/dist/learning/unattended-progress.js.map +1 -0
  99. package/dist/learning/working-state.d.ts +109 -0
  100. package/dist/learning/working-state.d.ts.map +1 -0
  101. package/dist/learning/working-state.js +244 -0
  102. package/dist/learning/working-state.js.map +1 -0
  103. package/dist/nlu/conversation-gate.d.ts +24 -0
  104. package/dist/nlu/conversation-gate.d.ts.map +1 -1
  105. package/dist/nlu/conversation-gate.js +47 -3
  106. package/dist/nlu/conversation-gate.js.map +1 -1
  107. package/dist/tools/coding-tools.d.ts.map +1 -1
  108. package/dist/tools/coding-tools.js +95 -19
  109. package/dist/tools/coding-tools.js.map +1 -1
  110. package/dist/tools/edit-verification.d.ts +125 -0
  111. package/dist/tools/edit-verification.d.ts.map +1 -0
  112. package/dist/tools/edit-verification.js +237 -0
  113. package/dist/tools/edit-verification.js.map +1 -0
  114. package/dist/tools/git-tool.d.ts.map +1 -1
  115. package/dist/tools/git-tool.js +37 -4
  116. package/dist/tools/git-tool.js.map +1 -1
  117. package/dist/tools/registry.d.ts +30 -2
  118. package/dist/tools/registry.d.ts.map +1 -1
  119. package/dist/tools/registry.js +40 -11
  120. package/dist/tools/registry.js.map +1 -1
  121. package/dist/tools/run-cli.d.ts.map +1 -1
  122. package/dist/tools/run-cli.js +40 -13
  123. package/dist/tools/run-cli.js.map +1 -1
  124. package/dist/tools/run-terminal.d.ts +6 -1
  125. package/dist/tools/run-terminal.d.ts.map +1 -1
  126. package/dist/tools/run-terminal.js +158 -18
  127. package/dist/tools/run-terminal.js.map +1 -1
  128. package/dist/tools/tool-loop.d.ts +35 -0
  129. package/dist/tools/tool-loop.d.ts.map +1 -1
  130. package/dist/tools/tool-loop.js +141 -1
  131. package/dist/tools/tool-loop.js.map +1 -1
  132. package/dist/web-dashboard/chat-console.d.ts +6 -0
  133. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  134. package/dist/web-dashboard/chat-console.js.map +1 -1
  135. package/dist/web-dashboard/chat-retry.d.ts.map +1 -1
  136. package/dist/web-dashboard/chat-retry.js +10 -2
  137. package/dist/web-dashboard/chat-retry.js.map +1 -1
  138. package/dist/web-dashboard/server.d.ts.map +1 -1
  139. package/dist/web-dashboard/server.js +80 -8
  140. package/dist/web-dashboard/server.js.map +1 -1
  141. package/dist/web-dashboard/src/types.d.ts +81 -4
  142. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  143. package/package.json +1 -1
  144. package/src/web-dashboard/public/assets/{index-Co7Hk2FT.js → index-kCUkORm7.js} +2 -2
  145. package/src/web-dashboard/public/assets/{index-Co7Hk2FT.js.map → index-kCUkORm7.js.map} +1 -1
  146. package/src/web-dashboard/public/index.html +1 -1
@@ -28,6 +28,11 @@ import { shouldPromptWeakModel, promptWeakModelChoice } from '../cli/weak-model-
28
28
  import { logger } from '../utils/logger.js';
29
29
  import { ContextVault } from './context-vault.js';
30
30
  import { saveCheckpoint, loadCheckpoint, checkpointIdFor } from './checkpoint-store.js';
31
+ import { buildLongFormPlan, isContinuationAsk, } from './long-form-plan.js';
32
+ import { buildCompositePlan } from './composite-plan.js';
33
+ import { assembleDocument, countWords, findInProgressJob, formatProgress, jobProgress, recordSectionOutcome, } from '../learning/long-form.js';
34
+ import { artifactsPresence } from '../learning/unattended-progress.js';
35
+ import { resolveDefaultProvider } from '../learning/model-selection.js';
31
36
  import { buildProjectFileTree, truncateTree, SOURCE_EXTENSIONS, IGNORE_DIRS } from './utils/file-tree.js';
32
37
  import { cleanupSandbox } from './agents/tester.js';
33
38
  import { getMCPManager, resetMCPManager } from '../mcp/manager.js';
@@ -63,11 +68,13 @@ import { classifyFallbackError, getProviderFallback, recordRegistrySuccess } fro
63
68
  import { recordActionFailure } from '../learning/failure-bookkeeping.js';
64
69
  import { sweepTransientFailures, sessionRevivalStore } from '../learning/provider-revival.js';
65
70
  import { resolveContextBudget, resolveContextFileBudget, resolveMaxOutputTokens } from '../learning/context-budget.js';
71
+ import { clampMaxTokens, learnMaxTokensLimitFromError } from '../learning/provider-limits.js';
66
72
  import { resolveWorkingModel } from '../inference/model-validator.js';
67
73
  import { refreshModelRegistry } from '../inference/model-probe.js';
68
74
  import { recordRoutingDecision } from '../learning/routing-history.js';
69
75
  import { getQuotaLedger } from '../learning/quota-ledger.js';
70
76
  import { withTraceCapture, beginTrace, endTrace } from '../learning/reasoning-trace.js';
77
+ import { recordWorkingState } from '../learning/working-state.js';
71
78
  import { createResilientCallLLM } from '../learning/resilient-call.js';
72
79
  import { createReviewFromResult } from '../team/review.js';
73
80
  import { indexFiles, retrieve, recordRetrievalStats, retrievalOptionsFromConfig, estimateTokens as retrievalEstimateTokens } from '../learning/retrieval.js';
@@ -96,6 +103,20 @@ async function tryUpdateDAGNode(nodeId, update) {
96
103
  async function tryResetDAG() {
97
104
  resetDAG();
98
105
  }
106
+ /**
107
+ * Include `pendingWork` only when there is genuinely something left to do.
108
+ *
109
+ * Extracted so the rule lives in one place: every surface treats a handed-over
110
+ * job as work to schedule, so handing over a COMPLETE deliverable would make it
111
+ * rebuild the thing the user already has.
112
+ */
113
+ function buildPendingWork(pending) {
114
+ if (!pending)
115
+ return {};
116
+ if (pending.percent >= 100)
117
+ return {};
118
+ return { pendingWork: pending };
119
+ }
99
120
  // ─── Constants ──────────────────────────────────────────────────────────────
100
121
  /** Valid per-subtask complexity labels (mirrors ComplexityLevel). */
101
122
  const VALID_COMPLEXITY = new Set(['trivial', 'simple', 'moderate', 'complex', 'critical']);
@@ -273,11 +294,17 @@ export class Orchestrator {
273
294
  // P0 reasoning trace: begin the per-pipeline trace so every planner,
274
295
  // memory, and task LLM call lands in ~/.nuvira/memory/reasoning-traces.json
275
296
  // (best-effort — a trace failure must never break the pipeline).
297
+ // Audit honesty: record the provider×model that will ACTUALLY serve the
298
+ // calls. `options.provider` is undefined whenever the user did not pin one
299
+ // (`nuvira run "…"` with no --provider), and the trace then labelled every
300
+ // step `provider: unknown` — including the writer steps of a live 5-page
301
+ // story run. A trace nobody can attribute is not an audit trail.
302
+ const auditRoute = this.resolveAuditRoute(options);
276
303
  this.activeTraceId = beginTrace({
277
304
  goal,
278
305
  source: 'orchestrator',
279
- provider: options.provider,
280
- model: options.model,
306
+ provider: auditRoute.provider,
307
+ model: auditRoute.model,
281
308
  });
282
309
  let result;
283
310
  try {
@@ -306,6 +333,23 @@ export class Orchestrator {
306
333
  catch {
307
334
  // Best-effort — never break the result delivery over telemetry.
308
335
  }
336
+ // Enterprise G3 — record what this run DID for the next one: the files it
337
+ // changed, whether a verification agent (tester/runner/verifier) ran, and
338
+ // the goal itself (scanned for a regression signal). Best-effort.
339
+ try {
340
+ const changed = result?.changedFiles ?? [];
341
+ const verificationRan = (result?.agentResults ?? []).some((r) => r.success && /test|runner|verif|audit/i.test(r.agent));
342
+ recordWorkingState(process.cwd(), {
343
+ filesTouched: changed,
344
+ toolsUsed: ['pipeline'],
345
+ verified: result?.success === true && verificationRan,
346
+ unverifiedEdit: changed.length > 0 && !(result?.success === true && verificationRan),
347
+ userMessage: goal,
348
+ });
349
+ }
350
+ catch {
351
+ // Best-effort — a ledger write must never break result delivery.
352
+ }
309
353
  this.activeTraceId = null;
310
354
  // K2: persist runtime metrics (memory hits/misses, rule/LLM latency)
311
355
  // at the end of every pipeline run so nuvira doctor / the dashboard see
@@ -794,36 +838,94 @@ export class Orchestrator {
794
838
  }
795
839
  }
796
840
  else {
841
+ // ── AUTHORED DELIVERABLES plan themselves — BEFORE the design layers ──
842
+ // The planner's output for "write a 100-page story" was a Python script
843
+ // that would write the story: ZERO prose steps, and the failing run died
844
+ // before even that script existed. Two conclusions follow, and both are
845
+ // why this is computed FIRST rather than after planning:
846
+ // 1. the unit plan IS the plan for an authored ask — the code planner
847
+ // has no vocabulary for "39 chapters", so its output is replaced;
848
+ // 2. a planner failure must not sink a story. The live session's six
849
+ // failures were all JSON/planning-layer deaths; a book does not need
850
+ // the planner to exist, so it is no longer a dependency of one.
851
+ //
852
+ // G14 — it also runs before the REASONER now, because whether this run is
853
+ // CONTINUING work that is already in flight decides whether the design
854
+ // layers run at all (see below).
855
+ const priorAuthoredJob = this.findInFlightAuthoredJob(vault);
856
+ let longFormPlan = null;
857
+ try {
858
+ longFormPlan = this.planAuthored(vault, goal, options);
859
+ }
860
+ catch (err) {
861
+ // Best-effort — a long-form planning failure must never break the run.
862
+ logger.debug(`long-form planning failure: ${err instanceof Error ? err.message : String(err)}`);
863
+ }
864
+ // ── G14: a CONTINUATION batch does not re-decide the design ──────────
865
+ // Continuing an in-flight authored job re-derives the unit/phase plan from
866
+ // the ledger and REPLACES whatever the reasoner and planner produce — the
867
+ // deliverable class is in the ledger, the structure is a deterministic
868
+ // function of it, and nothing about the design is still open. Running them
869
+ // anyway costs two LLM round trips per batch and, worse, adds a failure
870
+ // surface the plan does not need: live evidence from the unattended web-book
871
+ // run shows the planner failing with `provider-error` on EVERY batch from
872
+ // batch 4 onward and burning its entire repair budget before the real work
873
+ // started. Both layers are now skipped, and the skip is REPORTED rather
874
+ // than hidden — "no planner was needed" is a fact about the run.
875
+ //
876
+ // Scoped deliberately to CONTINUATIONS. A first, fresh authored ask still
877
+ // runs both: that is the one run where the class is being established, and
878
+ // it is a single batch, not one per batch.
879
+ const continuingAuthored = !!(longFormPlan && priorAuthoredJob);
797
880
  // ── 3d. Reasoner (technical decisions before planning) ─────────────
798
881
  // The reasoner makes high-level technical decisions (language, framework,
799
882
  // platform, architecture) BEFORE the planner creates steps. This replaces
800
883
  // generic "create a game" with specific "Create a Python+tkinter game,
801
884
  // single file, package with pyinstaller" — the planner then creates
802
885
  // precise steps based on these decisions.
803
- try {
804
- if (options.verbose)
805
- logger.highlight('\n🧠 Reasoning...');
806
- const reasoner = this.moduleRegistry.getModule('reasoner');
807
- const reasonerResult = await this.runAgent(reasoner, vault, plannerCallLLM, options);
808
- agentResults.push({ agent: 'Reasoner', success: reasonerResult.success, summary: reasonerResult.summary });
809
- if (options.verbose && reasonerResult.success) {
810
- const decision = vault.getMeta('technicalDecision');
811
- if (decision) {
812
- logger.info(` 🧠 ${decision.language}+${decision.framework} → ${decision.platform} → ${decision.deliverable}`);
813
- if (decision.reasoning)
814
- logger.info(` 🧠 ${decision.reasoning}`);
815
- }
886
+ if (continuingAuthored) {
887
+ agentResults.push({
888
+ agent: 'Reasoner',
889
+ success: true,
890
+ summary: 'Skipped — continuing in-flight authored work (the deliverable class is already decided)',
891
+ });
892
+ if (options.verbose) {
893
+ logger.info(' ⏭️ Continuing in-flight authored work — reasoner skipped');
816
894
  }
817
- // Best-effort — reasoning failure must never block planning
818
895
  }
819
- catch (err) {
820
- logger.debug(`Reasoner failed (non-critical): ${err}`);
896
+ else {
897
+ try {
898
+ if (options.verbose)
899
+ logger.highlight('\n🧠 Reasoning...');
900
+ const reasoner = this.moduleRegistry.getModule('reasoner');
901
+ const reasonerResult = await this.runAgent(reasoner, vault, plannerCallLLM, options);
902
+ agentResults.push({ agent: 'Reasoner', success: reasonerResult.success, summary: reasonerResult.summary });
903
+ if (options.verbose && reasonerResult.success) {
904
+ const decision = vault.getMeta('technicalDecision');
905
+ if (decision) {
906
+ logger.info(` 🧠 ${decision.language}+${decision.framework} → ${decision.platform} → ${decision.deliverable}`);
907
+ if (decision.reasoning)
908
+ logger.info(` 🧠 ${decision.reasoning}`);
909
+ }
910
+ }
911
+ // Best-effort — reasoning failure must never block planning
912
+ }
913
+ catch (err) {
914
+ logger.debug(`Reasoner failed (non-critical): ${err}`);
915
+ }
821
916
  }
822
- if (options.verbose)
917
+ if (options.verbose && !continuingAuthored)
823
918
  logger.highlight('\n📋 Planning...');
824
919
  // Planner with auto-repair — if planning fails, try alternative approaches
825
920
  // instead of immediately giving up with "Planning failed".
826
- let planResult = await this.runAgent(this.moduleRegistry.getModule('planner'), vault, plannerCallLLM, options);
921
+ // (Skipped on a continuation; the synthesized result is a RECORD of that,
922
+ // not a claim that planning succeeded — see the summary text.)
923
+ let planResult = continuingAuthored
924
+ ? {
925
+ success: true,
926
+ summary: 'Skipped — continuing in-flight authored work (the unit plan is the plan)',
927
+ }
928
+ : await this.runAgent(this.moduleRegistry.getModule('planner'), vault, plannerCallLLM, options);
827
929
  if (!planResult.success) {
828
930
  // MODEL ESCALATION on planner failure (assessment P0: "deliver the
829
931
  // goal, complete the task in iterations"). A planner failure is almost
@@ -882,7 +984,23 @@ export class Orchestrator {
882
984
  this.stats.recoveredFailures += 1;
883
985
  }
884
986
  agentResults.push({ agent: 'Planner', success: planResult.success, summary: planResult.summary });
885
- if (!planResult.success) {
987
+ if (longFormPlan) {
988
+ // The unit/phase plan IS the plan. Adopt it whether or not the code
989
+ // planner succeeded — and REPLACE a code plan when it did, because the
990
+ // planner CANNOT express this deliverable correctly (it asks for a
991
+ // script that would write the story instead of writing the story).
992
+ vault.context.taskPlan = longFormPlan.steps;
993
+ agentResults.push({
994
+ agent: longFormPlan.kind === 'phased' ? 'CompositePlanner' : 'LongFormPlanner',
995
+ success: true,
996
+ summary: `${longFormPlan.steps.length} step(s) planned${longFormPlan.resumed ? ' (resumed)' : ''} — ${longFormPlan.label}`,
997
+ });
998
+ logger.info(` ${longFormPlan.kind === 'phased' ? '🛠️' : '📖'} ${longFormPlan.kind === 'phased' ? 'Composite plan' : 'Long-form plan'}: ${longFormPlan.label}`);
999
+ if (!planResult.success) {
1000
+ logger.warn(' ⚠️ Planner failed — continuing with the content plan (no planner needed for an authored deliverable)');
1001
+ }
1002
+ }
1003
+ else if (!planResult.success) {
886
1004
  const errMsg = planResult.error || 'Planning failed';
887
1005
  // Provide actionable guidance based on the error type.
888
1006
  let hint = '';
@@ -1234,8 +1352,14 @@ export class Orchestrator {
1234
1352
  });
1235
1353
  // Format as text for the result summary
1236
1354
  const reportText = this.reportModule.format(report, 'text');
1355
+ // ── Long-form work: state what EXISTS, not just whether the batch ran ──
1356
+ // A 100-page book cannot finish in one run, so "success" for the batch is
1357
+ // NOT "the deliverable is done". This appends the honest progress line
1358
+ // (units written, words on disk, how to continue) that the original six
1359
+ // failing runs never produced.
1360
+ const longFormSummary = this.longFormProgressNote(vault);
1237
1361
  return this.buildResult(!hasFailures, goal, agentResults, vault, {
1238
- summary: reportText,
1362
+ summary: longFormSummary ? `${reportText}\n\n${longFormSummary}` : reportText,
1239
1363
  tasksCompleted: completed,
1240
1364
  tasksTotal: total,
1241
1365
  trajectoryId,
@@ -1533,9 +1657,15 @@ export class Orchestrator {
1533
1657
  // Output cap: explicit option → configured value → the model's real
1534
1658
  // capability. The old flat `4096` held a 200K+ model to a small-model
1535
1659
  // ceiling, so the better the model, the more of it the constant wasted.
1536
- maxTokens: inferenceOptions?.maxTokens ??
1660
+ //
1661
+ // Then clamped by anything the provider has TAUGHT us (G15): a model
1662
+ // with a large context window and a tiny output cap (the live story run
1663
+ // hit a 512-token cap on a big-window model) rejects the window-derived
1664
+ // number with a 400 every single call, and a hard-coded caller constant
1665
+ // like the prose path's 8192 cannot be allowed to do that silently.
1666
+ maxTokens: clampMaxTokens(inferenceOptions?.maxTokens ??
1537
1667
  config.maxTokens ??
1538
- resolveMaxOutputTokens({ provider: providerType, model: servedModel }),
1668
+ resolveMaxOutputTokens({ provider: providerType, model: servedModel }), providerType, servedModel),
1539
1669
  };
1540
1670
  // The strongest signal the provider×model is NOT usable: a real call
1541
1671
  // failed. Feed the SHARED registry telemetry path (the same one chat
@@ -1550,6 +1680,32 @@ export class Orchestrator {
1550
1680
  output = await provider.generate(prompt, mergedOptions);
1551
1681
  }
1552
1682
  catch (err) {
1683
+ // A provider that NAMES its output limit has told us what to do, and
1684
+ // retrying at that number is strictly better than failing the step
1685
+ // (G15 — the live story run burned 6 batches of prose units this way).
1686
+ // One retry, only when the named limit is actually below what we sent.
1687
+ const namedLimit = learnMaxTokensLimitFromError(err, providerType, servedModel);
1688
+ if (namedLimit !== null && mergedOptions.maxTokens !== undefined && mergedOptions.maxTokens > namedLimit) {
1689
+ logger.warn(`${providerType}/${servedModel} caps output at ${namedLimit} tokens (we sent ` +
1690
+ `${mergedOptions.maxTokens}) — clamping and retrying once`);
1691
+ try {
1692
+ output = await provider.generate(prompt, { ...mergedOptions, maxTokens: namedLimit });
1693
+ recordRegistrySuccess(providerType, mergedOptions.model, 'execute');
1694
+ this.stats.llmCalls += 1;
1695
+ this.stats.inputTokens += estimateTokens(prompt);
1696
+ this.stats.outputTokens += estimateTokens(output);
1697
+ return output;
1698
+ }
1699
+ catch (retryErr) {
1700
+ // The cap was not the (only) problem — surface the ORIGINAL error
1701
+ // with the retry attached, so the audit reads the first cause.
1702
+ recordActionFailure(this.failureSession, providerType, retryErr, this.configManager, {
1703
+ model: mergedOptions.model,
1704
+ action: 'execute',
1705
+ });
1706
+ throw retryErr;
1707
+ }
1708
+ }
1553
1709
  // FULL shared bookkeeping (Nuvira-Router M0.2 Stage C): the previous
1554
1710
  // bare recordRegistryFailure only updated health scores — a mid-pipeline
1555
1711
  // 429 now also parks the provider in the quota ledger (so the NEXT task
@@ -1894,6 +2050,11 @@ export class Orchestrator {
1894
2050
  // batches: writer/runner look up "the running task" in the shared plan, so
1895
2051
  // a per-task marker disambiguates when several run concurrently.
1896
2052
  vault.setMeta('currentTaskId', task.id);
2053
+ // Long-form unit? Hand the writer its unit brief (title, path, word target,
2054
+ // and the previous unit's tail for continuity). Cleared for every other
2055
+ // step so a code task can never inherit a prose contract — or vice versa.
2056
+ const proseEntry = this.proseUnits.get(task.id);
2057
+ vault.setMeta('proseUnit', proseEntry ? proseEntry.unit : undefined);
1897
2058
  await tryUpdateDAGNode(task.id, { status: 'running' });
1898
2059
  this.eventBus.emit(EventNames.ORCHESTRATOR_TASK_STARTED, {
1899
2060
  taskId: task.id,
@@ -1922,6 +2083,10 @@ export class Orchestrator {
1922
2083
  const agentModel = options.model || options.agentModels?.[effectiveAgentType] || options.agentModels?.[task.agentType];
1923
2084
  let taskBoundProvider;
1924
2085
  const useResilient = autoRouting && (options.resilientRouting !== false);
2086
+ // The provider×model that will actually serve this step's calls (see
2087
+ // `resolveAuditRoute`). Resolved ONCE per step so the trace records a
2088
+ // real attribution even when the user pinned nothing.
2089
+ const stepAuditRoute = this.resolveAuditRoute(options, agentModel);
1925
2090
  const routedAgentCallLLM = autoRouting
1926
2091
  ? (useResilient
1927
2092
  ? this.createResilientAutoRoutedLLM({
@@ -1947,8 +2112,8 @@ export class Orchestrator {
1947
2112
  agentType: effectiveAgentType,
1948
2113
  taskId: task.id,
1949
2114
  description: task.description,
1950
- provider: options.provider,
1951
- model: agentModel || options.model,
2115
+ provider: stepAuditRoute.provider,
2116
+ model: stepAuditRoute.model,
1952
2117
  });
1953
2118
  // ── AGENT ANSWER-QUALITY GATE ──────────────────────────────────────
1954
2119
  // The agents do NOT go through the loop engine, so the generation-time
@@ -1999,7 +2164,20 @@ export class Orchestrator {
1999
2164
  // of a single-shot LLM call.
2000
2165
  let actualAgentType = effectiveAgentType;
2001
2166
  if (resolveUseToolCalling(options)) {
2002
- if (effectiveAgentType === 'writer') {
2167
+ // A long-form PROSE unit must NOT go to the tool-calling writer: that
2168
+ // agent speaks a read→propose_change→edit protocol for code, and would
2169
+ // try to emit file changes from tool calls instead of writing prose.
2170
+ // The one-shot WriterAgent owns the author path (G7/G8).
2171
+ const isProseUnit = this.proseUnits.has(task.id);
2172
+ // Same for a step that CREATES a file in a project that does not exist
2173
+ // yet (a composite plan's scaffold/experience/services steps). The
2174
+ // tool-calling writer's value is the read→edit→verify loop, and there is
2175
+ // nothing to read on a greenfield. Live evidence: it returned ZERO file
2176
+ // changes for the site scaffold and the pipeline died on step 1 of 8 —
2177
+ // the whole composite deliverable unreachable for a reason that had
2178
+ // nothing to do with the plan.
2179
+ const isCreationStep = this.creationSteps.has(task.id);
2180
+ if (effectiveAgentType === 'writer' && !isProseUnit && !isCreationStep) {
2003
2181
  actualAgentType = 'writer-tc';
2004
2182
  }
2005
2183
  else if (effectiveAgentType === 'reviewer') {
@@ -2390,6 +2568,47 @@ export class Orchestrator {
2390
2568
  }
2391
2569
  }
2392
2570
  }
2571
+ // ── Long-form unit finished: measure it, record it, assemble when done ──
2572
+ // "The writer step succeeded" is not the same as "the unit exists", so
2573
+ // the outcome is measured from the FILE that landed on disk. The ledger
2574
+ // turns that into progress the next turn can resume from.
2575
+ const prose = this.proseUnits.get(task.id);
2576
+ if (prose) {
2577
+ const job = this.longFormJobs.get(prose.jobKey);
2578
+ // The unit brief carries the absolute path of its own file.
2579
+ const abs = prose.unit.absolutePath;
2580
+ let words = 0;
2581
+ try {
2582
+ if (existsSync(abs))
2583
+ words = countWords(readFileSync(abs, 'utf-8'));
2584
+ }
2585
+ catch {
2586
+ words = 0;
2587
+ }
2588
+ const ok = result.success && words > 0;
2589
+ if (job) {
2590
+ recordSectionOutcome(job, prose.unit.index, {
2591
+ ok,
2592
+ words,
2593
+ error: ok ? undefined : result.error || `unit produced ${words} words`,
2594
+ });
2595
+ this.longFormJobs.set(prose.jobKey, job);
2596
+ vault.setMeta('longFormJobKey', prose.jobKey);
2597
+ // Keep the hand-off honest: the surface schedules the NEXT batch from
2598
+ // this snapshot, so it must reflect the units just written.
2599
+ const prior = this.pendingWork;
2600
+ this.pendingWork = this.pendingWorkFor(job, prior?.kind ?? 'long-form', prior?.expectedArtifacts);
2601
+ if (options.verbose) {
2602
+ logger.info(` 📖 ${prose.unit.title}: ${words} words recorded`);
2603
+ }
2604
+ if (ok) {
2605
+ const assembled = assembleDocument(job);
2606
+ if (assembled) {
2607
+ logger.success(` 📖 Document assembled: ${assembled.path} — ${assembled.words.toLocaleString()} words (~${assembled.pages} pages, ${assembled.files} units)`);
2608
+ }
2609
+ }
2610
+ }
2611
+ }
2393
2612
  // After runner step: refresh artifacts with any files created during execution
2394
2613
  if (effectiveAgentType === 'runner' && result.success) {
2395
2614
  const runResult = vault.getMeta('runResult');
@@ -3009,6 +3228,19 @@ export class Orchestrator {
3009
3228
  * Fire-and-forget: never awaited, never blocks, never throws.
3010
3229
  */
3011
3230
  maybeFireColdStartProbe() {
3231
+ // ── The warmup daemon is NOT cold-start-only (Models-page audit) ───────
3232
+ // It used to start ONLY from the cold branch below, so from the first
3233
+ // verified model onward nothing ever warmed or verified anything again: the
3234
+ // routing pool froze at whatever the cold probe found, and then the
3235
+ // registry's own 7-day staleness rule began retiring models nothing
3236
+ // re-verified (observed: 12 verified models against 496 unverified, four of
3237
+ // the 12 within 24h of going stale).
3238
+ //
3239
+ // Starting it on every run is safe because it is unref'd (it can never hold
3240
+ // the process open) and its per-cycle budget is bounded. That unref is what
3241
+ // makes this possible — without it, starting the daemon on a normal run
3242
+ // would hang every CLI command forever.
3243
+ this.maybeStartWarmupDaemon();
3012
3244
  if (this.coldStartProbeFired)
3013
3245
  return;
3014
3246
  try {
@@ -3018,14 +3250,6 @@ export class Orchestrator {
3018
3250
  this.coldStartProbeFired = true;
3019
3251
  void refreshModelRegistry(this.configManager, { spotCheck: true }).then((result) => {
3020
3252
  logger.info(` 🌱 Cold-start registry probe: ${result.providersProbed.length} provider(s), ${result.verified} verified, ${result.unavailable} unavailable`);
3021
- // Start warmup daemon to keep frequently-used models warm
3022
- try {
3023
- const { startWarmupDaemon } = require('../learning/model-warmup.js');
3024
- startWarmupDaemon(this.configManager);
3025
- }
3026
- catch {
3027
- // Best-effort — warmup must never break routing
3028
- }
3029
3253
  }).catch(() => {
3030
3254
  // Best-effort — a failed probe must never break the pipeline.
3031
3255
  });
@@ -3034,6 +3258,21 @@ export class Orchestrator {
3034
3258
  // Best-effort.
3035
3259
  }
3036
3260
  }
3261
+ /** Whether this instance already started the background warmup daemon. */
3262
+ warmupDaemonStarted = false;
3263
+ /** Start the background model warmup/exploration daemon exactly once. */
3264
+ maybeStartWarmupDaemon() {
3265
+ if (this.warmupDaemonStarted)
3266
+ return;
3267
+ this.warmupDaemonStarted = true;
3268
+ try {
3269
+ const { startWarmupDaemon } = require('../learning/model-warmup.js');
3270
+ startWarmupDaemon(this.configManager);
3271
+ }
3272
+ catch {
3273
+ // Best-effort — warmup must never break routing
3274
+ }
3275
+ }
3037
3276
  applyRoutingPlanAdjustments(vault, routingContext) {
3038
3277
  if (!routingContext?.taskProfile?.requiresVerification) {
3039
3278
  return;
@@ -3124,6 +3363,221 @@ export class Orchestrator {
3124
3363
  }
3125
3364
  return count;
3126
3365
  }
3366
+ /** Long-form jobs in flight this run, keyed by job key. */
3367
+ longFormJobs = new Map();
3368
+ /**
3369
+ * Enterprise G11 — the work this run leaves unfinished, handed to the calling
3370
+ * surface so it can keep going without a "continue" turn.
3371
+ */
3372
+ pendingWork;
3373
+ /**
3374
+ * The config's effective provider/model, resolved once per run.
3375
+ *
3376
+ * Only used for ATTRIBUTION (traces, events, review bundles). When the user
3377
+ * pinned a provider or a model those win; otherwise the running default is
3378
+ * looked up, so a trace says `gemini · gemini-2.0-flash` instead of
3379
+ * `unknown · unknown`. A live 5-page story run recorded every step as
3380
+ * unknown — the audit trail could not say which model wrote the book.
3381
+ */
3382
+ defaultRoute = null;
3383
+ resolveAuditRoute(options, agentModel) {
3384
+ const model = agentModel || options.model;
3385
+ if (options.provider && model)
3386
+ return { provider: options.provider, model };
3387
+ if (!this.defaultRoute) {
3388
+ let provider;
3389
+ let configuredModel;
3390
+ try {
3391
+ provider = resolveDefaultProvider(this.configManager);
3392
+ configuredModel = this.configManager.getProviderConfig(provider).config?.model;
3393
+ }
3394
+ catch {
3395
+ // Best-effort attribution — an unresolvable default stays undefined
3396
+ // and the trace falls back to its own 'unknown' marker.
3397
+ }
3398
+ this.defaultRoute = { provider, model: configuredModel };
3399
+ }
3400
+ return { provider: options.provider ?? this.defaultRoute.provider, model: model ?? this.defaultRoute.model };
3401
+ }
3402
+ /** The prose unit + owning job for each long-form task id. */
3403
+ proseUnits = new Map();
3404
+ /** Steps that create files from scratch, keyed by task id (one-shot writer). */
3405
+ creationSteps = new Set();
3406
+ /**
3407
+ * Plan an authored ask as bounded prose units.
3408
+ *
3409
+ * Returns null for every non-authored goal, so the ordinary code path is
3410
+ * untouched. For an authored goal whose units are already all written, the
3411
+ * finished document is assembled and no work is planned — a "continue" turn
3412
+ * after completion must not restart the book.
3413
+ */
3414
+ planAuthored(vault, goal, options) {
3415
+ const workingDir = vault.context.workingDirectory || process.cwd();
3416
+ const forceResume = isContinuationAsk(goal);
3417
+ // ── Hybrid asks get PHASES, not units ──────────────────────────────────
3418
+ // "a web-based interactive book with voice" is authored content AND an
3419
+ // application. Planning it as either alone loses half the ask, so it is
3420
+ // planned as ordered phases (shape → content → experience → services →
3421
+ // verify) with a deterministic verification step at the end.
3422
+ let composite = null;
3423
+ try {
3424
+ composite = buildCompositePlan({ goal, workingDir });
3425
+ }
3426
+ catch (err) {
3427
+ logger.debug(`composite planning skipped: ${err instanceof Error ? err.message : String(err)}`);
3428
+ }
3429
+ if (composite) {
3430
+ const presence = artifactsPresence(workingDir, composite.expectedArtifacts);
3431
+ const progress = jobProgress(composite.job);
3432
+ // Everything already on disk? Then there is nothing to plan. (A re-run of
3433
+ // a finished ask must never rebuild the shell over finished chapters.)
3434
+ if (progress.complete && presence.present === presence.total) {
3435
+ const assembled = assembleDocument(composite.job);
3436
+ if (assembled) {
3437
+ logger.success(` 🛠️ Deliverable already complete — assembled ${assembled.path}`);
3438
+ }
3439
+ return null;
3440
+ }
3441
+ this.longFormJobs.set(composite.job.key, composite.job);
3442
+ for (const [stepId, unit] of composite.proseUnits) {
3443
+ this.proseUnits.set(stepId, { unit, jobKey: composite.job.key });
3444
+ }
3445
+ // Greenfield creation steps go to the one-shot writer (see the exception
3446
+ // in `runAgent`): a read→edit loop has nothing to read in a directory that
3447
+ // does not exist yet.
3448
+ for (const id of composite.creationStepIds)
3449
+ this.creationSteps.add(id);
3450
+ vault.setMeta('longFormJobKey', composite.job.key);
3451
+ vault.setMeta('compositeArtifacts', composite.expectedArtifacts);
3452
+ const plan = {
3453
+ steps: composite.steps,
3454
+ progressLine: composite.progressLine,
3455
+ resumed: forceResume,
3456
+ kind: 'phased',
3457
+ label: `${composite.deliverableSummary} — ${composite.progressLine}`,
3458
+ expectedArtifacts: composite.expectedArtifacts,
3459
+ job: composite.job,
3460
+ composite,
3461
+ };
3462
+ this.pendingWork = this.pendingWorkFor(composite.job, 'phased', composite.expectedArtifacts);
3463
+ if (options.verbose) {
3464
+ logger.info(` 🛠️ ${composite.substrates.join(' + ')} → ${composite.phases.map((p) => p.title).join(' → ')}`);
3465
+ }
3466
+ return plan;
3467
+ }
3468
+ // ── Everything else: the single-substrate unit plan (G8) ───────────────
3469
+ // A bare "continue" carries no class or magnitude — the ledger does, so it
3470
+ // resumes the project's in-flight job instead of starting a second book
3471
+ // from the default filename.
3472
+ const plan = buildLongFormPlan({ goal, workingDir, forceResume });
3473
+ if (!plan)
3474
+ return null;
3475
+ if (plan.steps.length === 0) {
3476
+ const assembled = assembleDocument(plan.job);
3477
+ if (assembled) {
3478
+ logger.success(` 📖 Nothing left to write — assembled ${assembled.path} (${assembled.words.toLocaleString()} words, ~${assembled.pages} pages)`);
3479
+ }
3480
+ return null;
3481
+ }
3482
+ this.longFormJobs.set(plan.job.key, plan.job);
3483
+ for (const [stepId, unit] of plan.units) {
3484
+ this.proseUnits.set(stepId, { unit, jobKey: plan.job.key });
3485
+ }
3486
+ vault.setMeta('longFormJobKey', plan.job.key);
3487
+ const authored = {
3488
+ steps: plan.steps,
3489
+ progressLine: plan.progressLine,
3490
+ resumed: plan.resumed,
3491
+ kind: 'long-form',
3492
+ label: plan.progressLine,
3493
+ job: plan.job,
3494
+ };
3495
+ this.pendingWork = this.pendingWorkFor(plan.job, 'long-form');
3496
+ if (options.verbose && plan.resumed) {
3497
+ logger.info(` 📖 Resuming long-form work — ${plan.progressLine}`);
3498
+ }
3499
+ return authored;
3500
+ }
3501
+ /**
3502
+ * The in-flight authored job for this project, if any (G14).
3503
+ *
3504
+ * The signal is the LEDGER, not the wording of the ask. A composite
3505
+ * continuation re-sends the ORIGINAL goal — its phases are re-derived against
3506
+ * the ledger — so `isContinuationAsk('continue')` would miss it entirely. An
3507
+ * in-flight job for this project IS the continuation, whichever surface
3508
+ * scheduled it.
3509
+ */
3510
+ findInFlightAuthoredJob(vault) {
3511
+ try {
3512
+ const workingDir = vault.context.workingDirectory || process.cwd();
3513
+ return findInProgressJob(workingDir);
3514
+ }
3515
+ catch {
3516
+ return null;
3517
+ }
3518
+ }
3519
+ /**
3520
+ * Describe what is still outstanding, so the calling SURFACE can keep it
3521
+ * going without the user typing "continue".
3522
+ *
3523
+ * Measured from the ledger, never from the run's own success flag: a batch
3524
+ * that succeeded is 4 chapters of a 39-chapter book, and reporting that as
3525
+ * "done" is the overclaim this whole workstream exists to remove.
3526
+ */
3527
+ pendingWorkFor(job, kind, expectedArtifacts) {
3528
+ const progress = jobProgress(job);
3529
+ const presence = artifactsPresence(job.projectPath, expectedArtifacts);
3530
+ const unitsLeft = progress.total - progress.done;
3531
+ const filesLeft = presence.total - presence.present;
3532
+ const complete = progress.complete && filesLeft === 0;
3533
+ return {
3534
+ kind,
3535
+ goal: job.goal,
3536
+ projectPath: job.projectPath,
3537
+ // A composite deliverable is re-planned from the ORIGINAL ask (its phases
3538
+ // are re-derived against the ledger); a plain book resumes on "continue".
3539
+ continuationPrompt: kind === 'phased' ? job.goal : 'continue',
3540
+ ...(expectedArtifacts ? { expectedArtifacts } : {}),
3541
+ progressLine: formatProgress(job),
3542
+ percent: complete ? 100 : progress.percent,
3543
+ reason: complete
3544
+ ? 'the content is complete but the deliverable has not been assembled yet'
3545
+ : unitsLeft > 0
3546
+ ? `${unitsLeft} of ${progress.total} content units remaining`
3547
+ : `${filesLeft} deliverable file(s) remaining`,
3548
+ };
3549
+ }
3550
+ /**
3551
+ * The honest progress line for the pipeline summary.
3552
+ *
3553
+ * Reported even when the batch SUCCEEDED, because for a 39-unit book a
3554
+ * successful batch is 4 chapters — presenting that as "done" (or as a bare
3555
+ * "success") is the kind of overclaim this hardening exists to remove.
3556
+ *
3557
+ * It no longer asks the user to reply: the work is scheduled to continue on
3558
+ * its own (the surface reads `result.pendingWork`), so demanding a "continue"
3559
+ * would be asking for permission the ask already granted.
3560
+ */
3561
+ longFormProgressNote(vault) {
3562
+ const key = vault.getMeta('longFormJobKey');
3563
+ if (!key)
3564
+ return '';
3565
+ const job = this.longFormJobs.get(key);
3566
+ if (!job)
3567
+ return '';
3568
+ const pending = this.pendingWork;
3569
+ const progress = jobProgress(job);
3570
+ if (progress.complete) {
3571
+ const presence = artifactsPresence(job.projectPath, pending?.expectedArtifacts);
3572
+ if (presence.total > 0 && presence.present < presence.total) {
3573
+ return (`✅ All ${progress.total} content units written — ${progress.words.toLocaleString()} words. ` +
3574
+ `Still to produce: ${presence.missing.join(', ')}. Continuing automatically.`);
3575
+ }
3576
+ return `✅ All ${progress.total} units written — ${progress.words.toLocaleString()} words (~${progress.pagesTarget} pages).`;
3577
+ }
3578
+ return (`${formatProgress(job)} — continuing automatically; ` +
3579
+ `${progress.total - progress.done} unit(s) left, no reply needed.`);
3580
+ }
3127
3581
  buildResult(success, goal, agentResults, vault, overrides = {}) {
3128
3582
  const completed = overrides.tasksCompleted ?? agentResults.filter((r) => r.success).length;
3129
3583
  const total = overrides.tasksTotal ?? agentResults.length;
@@ -3135,11 +3589,23 @@ export class Orchestrator {
3135
3589
  tasksTotal: total,
3136
3590
  agentResults,
3137
3591
  fileChanges: vault.getDiffSummary(),
3592
+ // Enterprise G3 — the changed paths for the working-state ledger.
3593
+ changedFiles: overrides.changedFiles ??
3594
+ vault.context.fileChanges.filter((c) => c.status !== 'deleted').map((c) => c.path),
3138
3595
  runOutput: overrides.runOutput,
3139
3596
  error: overrides.error,
3140
3597
  trajectoryId: overrides.trajectoryId,
3141
3598
  reviewId: overrides.reviewId,
3142
3599
  stats: overrides.stats ?? this.stats,
3600
+ // Enterprise G11 — unfinished work, so the surface can continue it
3601
+ // unattended instead of asking the user to say "continue".
3602
+ //
3603
+ // Present only while work REMAINS, which is the whole contract: a surface
3604
+ // schedules whatever it is handed, so reporting 100%-complete work as
3605
+ // pending would make the runner re-run a finished deliverable. `percent`
3606
+ // is 100 exactly when the content is complete AND every promised artifact
3607
+ // exists (see `pendingWorkFor`), so that is the test.
3608
+ ...(buildPendingWork(overrides.pendingWork ?? this.pendingWork)),
3143
3609
  };
3144
3610
  }
3145
3611
  }