agent-nuvira 3.3.11 → 3.3.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. package/README.md +32 -1
  2. package/dist/agents/orchestrator.d.ts.map +1 -1
  3. package/dist/agents/orchestrator.js +28 -12
  4. package/dist/agents/orchestrator.js.map +1 -1
  5. package/dist/cli/chat.d.ts +36 -0
  6. package/dist/cli/chat.d.ts.map +1 -1
  7. package/dist/cli/chat.js +422 -36
  8. package/dist/cli/chat.js.map +1 -1
  9. package/dist/cli/config.d.ts.map +1 -1
  10. package/dist/cli/config.js +14 -1
  11. package/dist/cli/config.js.map +1 -1
  12. package/dist/cli/execute.d.ts.map +1 -1
  13. package/dist/cli/execute.js +8 -0
  14. package/dist/cli/execute.js.map +1 -1
  15. package/dist/cli/loop-executor.d.ts.map +1 -1
  16. package/dist/cli/loop-executor.js +12 -7
  17. package/dist/cli/loop-executor.js.map +1 -1
  18. package/dist/cli/tool-install-prompt.d.ts +58 -0
  19. package/dist/cli/tool-install-prompt.d.ts.map +1 -1
  20. package/dist/cli/tool-install-prompt.js +137 -0
  21. package/dist/cli/tool-install-prompt.js.map +1 -1
  22. package/dist/cli/weak-model-prompt.d.ts +16 -7
  23. package/dist/cli/weak-model-prompt.d.ts.map +1 -1
  24. package/dist/cli/weak-model-prompt.js +24 -8
  25. package/dist/cli/weak-model-prompt.js.map +1 -1
  26. package/dist/config/types.d.ts +36 -0
  27. package/dist/config/types.d.ts.map +1 -1
  28. package/dist/context/cache.d.ts +35 -1
  29. package/dist/context/cache.d.ts.map +1 -1
  30. package/dist/context/cache.js +31 -2
  31. package/dist/context/cache.js.map +1 -1
  32. package/dist/inference/model-catalog.d.ts +17 -0
  33. package/dist/inference/model-catalog.d.ts.map +1 -1
  34. package/dist/inference/model-catalog.js +42 -7
  35. package/dist/inference/model-catalog.js.map +1 -1
  36. package/dist/inference/tool-call-utils.d.ts +4 -3
  37. package/dist/inference/tool-call-utils.d.ts.map +1 -1
  38. package/dist/inference/tool-call-utils.js +16 -4
  39. package/dist/inference/tool-call-utils.js.map +1 -1
  40. package/dist/learning/agentic-route-gate.d.ts +104 -0
  41. package/dist/learning/agentic-route-gate.d.ts.map +1 -0
  42. package/dist/learning/agentic-route-gate.js +125 -0
  43. package/dist/learning/agentic-route-gate.js.map +1 -0
  44. package/dist/learning/auto-router.d.ts +36 -0
  45. package/dist/learning/auto-router.d.ts.map +1 -1
  46. package/dist/learning/auto-router.js +115 -8
  47. package/dist/learning/auto-router.js.map +1 -1
  48. package/dist/learning/build-prerequisites.d.ts +68 -0
  49. package/dist/learning/build-prerequisites.d.ts.map +1 -0
  50. package/dist/learning/build-prerequisites.js +267 -0
  51. package/dist/learning/build-prerequisites.js.map +1 -0
  52. package/dist/learning/continuation-intent.d.ts +43 -0
  53. package/dist/learning/continuation-intent.d.ts.map +1 -0
  54. package/dist/learning/continuation-intent.js +82 -0
  55. package/dist/learning/continuation-intent.js.map +1 -0
  56. package/dist/learning/model-harness.d.ts +19 -0
  57. package/dist/learning/model-harness.d.ts.map +1 -1
  58. package/dist/learning/model-harness.js +27 -0
  59. package/dist/learning/model-harness.js.map +1 -1
  60. package/dist/learning/prompt-budget.d.ts +91 -0
  61. package/dist/learning/prompt-budget.d.ts.map +1 -0
  62. package/dist/learning/prompt-budget.js +99 -0
  63. package/dist/learning/prompt-budget.js.map +1 -0
  64. package/dist/learning/reasoning-trace.d.ts +55 -2
  65. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  66. package/dist/learning/reasoning-trace.js +48 -0
  67. package/dist/learning/reasoning-trace.js.map +1 -1
  68. package/dist/learning/resilient-call.d.ts.map +1 -1
  69. package/dist/learning/resilient-call.js +16 -3
  70. package/dist/learning/resilient-call.js.map +1 -1
  71. package/dist/learning/routing-history.d.ts +13 -0
  72. package/dist/learning/routing-history.d.ts.map +1 -1
  73. package/dist/learning/routing-history.js.map +1 -1
  74. package/dist/learning/turn-report.d.ts +78 -0
  75. package/dist/learning/turn-report.d.ts.map +1 -0
  76. package/dist/learning/turn-report.js +140 -0
  77. package/dist/learning/turn-report.js.map +1 -0
  78. package/dist/tools/edit-verification.d.ts +2 -11
  79. package/dist/tools/edit-verification.d.ts.map +1 -1
  80. package/dist/tools/edit-verification.js +33 -1
  81. package/dist/tools/edit-verification.js.map +1 -1
  82. package/dist/tools/loop-skill-hint.d.ts +51 -0
  83. package/dist/tools/loop-skill-hint.d.ts.map +1 -1
  84. package/dist/tools/loop-skill-hint.js +87 -2
  85. package/dist/tools/loop-skill-hint.js.map +1 -1
  86. package/dist/tools/plan-store.d.ts +104 -3
  87. package/dist/tools/plan-store.d.ts.map +1 -1
  88. package/dist/tools/plan-store.js +250 -6
  89. package/dist/tools/plan-store.js.map +1 -1
  90. package/dist/tools/registry.d.ts +20 -1
  91. package/dist/tools/registry.d.ts.map +1 -1
  92. package/dist/tools/registry.js +17 -4
  93. package/dist/tools/registry.js.map +1 -1
  94. package/dist/tools/run-terminal.d.ts.map +1 -1
  95. package/dist/tools/run-terminal.js +34 -0
  96. package/dist/tools/run-terminal.js.map +1 -1
  97. package/dist/tools/tool-loop.d.ts +29 -0
  98. package/dist/tools/tool-loop.d.ts.map +1 -1
  99. package/dist/tools/tool-loop.js +259 -8
  100. package/dist/tools/tool-loop.js.map +1 -1
  101. package/dist/utils/effect-verification.js +1 -1
  102. package/dist/utils/effect-verification.js.map +1 -1
  103. package/dist/web-dashboard/chat-console.d.ts +21 -1
  104. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  105. package/dist/web-dashboard/chat-console.js +33 -5
  106. package/dist/web-dashboard/chat-console.js.map +1 -1
  107. package/dist/web-dashboard/server.d.ts.map +1 -1
  108. package/dist/web-dashboard/server.js +38 -0
  109. package/dist/web-dashboard/server.js.map +1 -1
  110. package/dist/web-dashboard/src/types.d.ts +71 -0
  111. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  112. package/package.json +4 -1
  113. package/src/web-dashboard/public/assets/index-BrZcYhy6.js +207 -0
  114. package/src/web-dashboard/public/assets/index-BrZcYhy6.js.map +1 -0
  115. package/src/web-dashboard/public/assets/{index-C9eBskc9.css → index-DsWczTa6.css} +1 -1
  116. package/src/web-dashboard/public/index.html +2 -2
  117. package/src/web-dashboard/public/assets/index-BdKf5Xw2.js +0 -207
  118. package/src/web-dashboard/public/assets/index-BdKf5Xw2.js.map +0 -1
package/dist/cli/chat.js CHANGED
@@ -27,6 +27,7 @@ import { getProviderFallback, classifyFallbackError, isRetryableError, isTransie
27
27
  import { recordActionFailure } from '../learning/failure-bookkeeping.js';
28
28
  import { resolveThreadBudgetChars } from '../learning/context-budget.js';
29
29
  import { getAutoRouter, isAutoModel, isAutoProvider, governanceVerdict } from '../learning/auto-router.js';
30
+ import { continuationSoftwareText } from '../learning/continuation-intent.js';
30
31
  import { estimateTokens } from '../learning/cost-tracker.js';
31
32
  import { getModelRegistry } from '../learning/model-registry.js';
32
33
  import { refreshModelRegistry } from '../inference/model-probe.js';
@@ -34,9 +35,9 @@ import { startWarmupDaemon } from '../learning/model-warmup.js';
34
35
  import { recordRoutingDecision } from '../learning/routing-history.js';
35
36
  import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
36
37
  import { buildAutoResolveOptions } from '../learning/resolve-options.js';
37
- import { buildDeepFailoverPool, createFailoverExclusionFilter } from '../learning/resilient-call.js';
38
+ import { buildDeepFailoverPool, createFailoverExclusionFilter, modelBreadthReport } from '../learning/resilient-call.js';
38
39
  import { parseRequestSync } from '../nlu/parser.js';
39
- import { PlanStore } from '../tools/plan-store.js';
40
+ import { createPersistentPlanStore } from '../tools/plan-store.js';
40
41
  import { withLogCorrelation } from '../enterprise/log.js';
41
42
  import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
42
43
  import { resolveDispatch } from '../nlu/actions.js';
@@ -45,10 +46,13 @@ import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
45
46
  // WS3 (#25) — the turn as a span, when an operator has asked for OTLP export.
46
47
  import { flushSpans, otelNoticeOnce, startTurnSpan } from '../observability/otel.js';
47
48
  import { detectAnswerQualityFailure, answerQualityError, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
48
- import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, buildTraceOutcome, traceOutcomeSucceeded } from '../learning/reasoning-trace.js';
49
+ import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, recordTurnReport, buildTraceOutcome, traceOutcomeSucceeded } from '../learning/reasoning-trace.js';
49
50
  import { recordWorkingState, getWorkingState, formatWorkingState, isProjectLedgerDir } from '../learning/working-state.js';
50
51
  import { getLoopExposureMode } from '../tools/toolsets.js';
51
- import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
52
+ import { resolveModelHarnessProfile, shouldSkipNativeTools, isAgenticCapableModel } from '../learning/model-harness.js';
53
+ import { assertAgenticRoute, setWeakModelConsent, resolveWeakModelPolicy, weakRouteNotice } from '../learning/agentic-route-gate.js';
54
+ import { resolvePromptBudget, measurePromptBudget, formatPromptBudgetBreakdown } from '../learning/prompt-budget.js';
55
+ import { buildTurnReport, formatTurnReport } from '../learning/turn-report.js';
52
56
  import { resolveAdapterDefault, hasCredentials } from '../learning/model-selection.js';
53
57
  import { buildLoopProjectContext } from '../tools/loop-project-context.js';
54
58
  import { sweepTransientFailures, collectionRevivalStore } from '../learning/provider-revival.js';
@@ -281,7 +285,12 @@ export async function generateWithTransientRetry(attempt, signal, onRetry) {
281
285
  }
282
286
  }
283
287
  }
284
- function buildToolSystemPrompt(parsed) {
288
+ /**
289
+ * Exported for the release gate (`tests/release/agent-contracts.test.ts`): the
290
+ * assembled system prompt's SIZE and its required contract clauses are a
291
+ * release invariant, not an implementation detail. See the gate for why.
292
+ */
293
+ export function buildToolSystemPrompt(parsed) {
285
294
  return [
286
295
  "You are Nuvira, Agent-Nuvira's AI agent. You code, create, write, analyze, and automate — anything the user needs. You identify as Nuvira (never 'Buff').",
287
296
  'Be precise and honest. When a request is ambiguous or incomplete, clarify with ask_user instead of guessing.',
@@ -316,9 +325,43 @@ function buildToolSystemPrompt(parsed) {
316
325
  '- Never ask the user to attach a folder, and never tell them to move files into a directory. If a request genuinely needs a workspace and none is attached, the app asks for one on its own — so just answer as best you can.',
317
326
  '- Treat a request as project work ONLY when it actually concerns the codebase or produces a file on disk. A general question that happens to contain a word like "class", "project", "test" or "document" is still a general question.',
318
327
  '',
328
+ // WHY THIS BLOCK EXISTS. Asked to build a Tauri app, a model ran
329
+ // `cargo --version` → exit 127 and then answered "I cannot install
330
+ // system-level software like Rust on your host machine — I am physically
331
+ // unable to build the Tauri application", handing the user a manual
332
+ // `curl … | sh` step. It was not sandboxed at all; it had a real shell and
333
+ // the user had just granted terminal permission (trace
334
+ // trace-1791127992452-qzgodi). A missing prerequisite is a step to
335
+ // complete, not a capability limit — so this states that plainly.
336
+ '## Toolchains and missing prerequisites',
337
+ '- You run on the user\u2019s REAL machine with a REAL shell (run_terminal). You are NOT sandboxed and NOT "enclosed".',
338
+ '- If a command fails because a tool is missing (`command not found`, exit 127), INSTALL it yourself with run_terminal — e.g. `brew install <tool>`, `xcode-select --install`, the language installer (rustup for cargo/rust, nvm/npm for node, pyenv/pip for python, go, …), or apt/winget — then re-run the original command.',
339
+ '- NEVER answer that you are unable or not permitted to install a prerequisite, and never hand the user an install step you were asked to do. A missing tool is a step to complete, not a wall.',
340
+ '',
319
341
  TOOL_CONTRACT_JSON,
320
342
  ].join('\n');
321
343
  }
344
+ /**
345
+ * #30 — does this turn carry an HONESTY FLAG?
346
+ *
347
+ * A flagged answer must NEVER be written to the shared response cache. The cache
348
+ * stores text only, so a later identical prompt — on the CLI, the dashboard or the
349
+ * gateway — was served the reply with every flag absent: a known-unverified claim
350
+ * (`unverifiedActionClaim` / `unverifiedEditClaim` / `unverifiedBuildClaim`), an
351
+ * announced-but-unperformed action (`unfulfilledPromise`), a missing deliverable
352
+ * (`undeliveredArtifact`) or an inert turn (`noActionTaken`) replayed as a clean
353
+ * answer, on every surface at once. The flags exist because those answers must
354
+ * not be replayed as settled, so the turn is not cached and re-derives instead.
355
+ */
356
+ export function turnCarriesHonestyFlag(result) {
357
+ return Boolean(result.unverifiedActionClaim ||
358
+ result.unfulfilledPromise ||
359
+ result.undeliveredArtifact ||
360
+ result.unverifiedBuildClaim ||
361
+ result.unverifiedEdit ||
362
+ result.unverifiedEditClaim ||
363
+ result.noActionTaken);
364
+ }
322
365
  // ─── ChatCommand ────────────────────────────────────────────────────────────
323
366
  export class ChatCommand extends BaseCommand {
324
367
  devModeAuto = false;
@@ -364,7 +407,7 @@ export class ChatCommand extends BaseCommand {
364
407
  * console injects a per-session store instead; this is the CLI/execute
365
408
  * default so a plan survives across turns within one chat session).
366
409
  */
367
- planStore = new PlanStore();
410
+ planStore = createPersistentPlanStore(`cli:${process.cwd()}`);
368
411
  /**
369
412
  * Whether the cold-start probe has fired this session. On a fresh registry
370
413
  * (no verified models yet) the FIRST auto pick fires a background
@@ -440,11 +483,27 @@ export class ChatCommand extends BaseCommand {
440
483
  ? await this.getProvider({})
441
484
  : await this.getProvider(mergedOpts);
442
485
  let model = mergedOpts.model;
486
+ // B — captured from the route so the post-route capability gate below can
487
+ // judge the FINAL pair without re-resolving anything.
488
+ let routeVerdict;
489
+ let routedText;
443
490
  if (autoMode) {
444
- const routed = await this.routeMessageAuto(message);
491
+ // Intent-aware escalation: a bare "yes"/"do it" continuing software work
492
+ // routes on the prior ask, not on the signal-free continuation. When
493
+ // there is no such hint the call is byte-identical to before.
494
+ const routingText = continuationSoftwareText(message, opts.history ?? []) ?? undefined;
495
+ const routed = routingText
496
+ ? await this.routeMessageAuto(message, [], { routingText })
497
+ : await this.routeMessageAuto(message);
445
498
  type = routed.type;
446
499
  provider = routed.provider;
447
500
  model = routed.model;
501
+ routedText = routingText;
502
+ routeVerdict = {
503
+ complexity: routed.complexity,
504
+ agenticCapable: routed.agenticCapable,
505
+ taskProfile: routed.taskProfile,
506
+ };
448
507
  }
449
508
  // P3 — tell the GUI where the turn is headed before the tool loop runs.
450
509
  const isLocalFallback = autoMode && type === 'local';
@@ -452,6 +511,85 @@ export class ChatCommand extends BaseCommand {
452
511
  ? ' ⚠️ local model only — run `nuvira models` or `nuvira provider set` to add a cloud provider'
453
512
  : '';
454
513
  opts.onProgress?.(` 🧠 routed to ${provider.name}${model ? ` / ${model}` : ''} — working…${localWarning}`);
514
+ // ── Workstream B — agentic capability gate (consent-first) ──────────────
515
+ // The router RECORDED a verdict (A1); the turn must ACT on it. A software/
516
+ // agentic ask must never silently run on a weak model. Consent is per
517
+ // SESSION: one answer covers this session, and a new chat/task asks again.
518
+ //
519
+ // Interactive = an injected askUser (the dashboard console) or a real TTY.
520
+ // A piped/headless run never reaches the ask — it falls to the configured
521
+ // policy, whose default (`deny`/retry) can never silently downgrade.
522
+ if (autoMode && routeVerdict && routeVerdict.agenticCapable === false) {
523
+ const sessionId = opts.debugSession;
524
+ const interactive = Boolean(opts.askUser) || Boolean(process.stdin.isTTY);
525
+ const policy = resolveWeakModelPolicy(this.configManager);
526
+ const asDecision = {
527
+ complexity: routeVerdict.complexity,
528
+ taskProfile: (routeVerdict.taskProfile ?? {
529
+ intent: 'unknown',
530
+ requiresVerification: false,
531
+ }),
532
+ provider: type,
533
+ model: model ?? '',
534
+ agenticCapable: false,
535
+ };
536
+ let gate = assertAgenticRoute(asDecision, { sessionId, policy, interactive });
537
+ if (gate.action === 'ask') {
538
+ try {
539
+ const ask = opts.askUser ??
540
+ (await import('../tools/ask-user.js')).renderAskUser;
541
+ const answer = await ask(`This is a software task, but only a weak model is available ` +
542
+ `(${provider.name}${model ? ` / ${model}` : ''}). How should I proceed for this session?`, [
543
+ { label: 'Approve the weak model for this session' },
544
+ { label: 'Wait for a strong model only' },
545
+ ], false);
546
+ const granted = Number(answer.index) === 0;
547
+ if (sessionId)
548
+ setWeakModelConsent(sessionId, granted ? 'granted' : 'denied');
549
+ gate = { ...gate, action: granted ? 'proceed-weak-consented' : 'retry-strong' };
550
+ }
551
+ catch {
552
+ // No answer reachable — treat as deny (never a silent downgrade).
553
+ gate = { ...gate, action: 'retry-strong' };
554
+ }
555
+ }
556
+ if (gate.action === 'retry-strong') {
557
+ // Try ONE re-route that EXCLUDES the weak provider; accept it only if
558
+ // the returned pair is genuinely agentic-capable. Nothing capable →
559
+ // refuse honestly rather than run the weak model against consent.
560
+ let next = null;
561
+ try {
562
+ next = routedText
563
+ ? await this.routeMessageAuto(message, [type], { routingText: routedText })
564
+ : await this.routeMessageAuto(message, [type]);
565
+ }
566
+ catch {
567
+ next = null;
568
+ }
569
+ if (next && isAgenticCapableModel(next.model, next.type)) {
570
+ type = next.type;
571
+ provider = next.provider;
572
+ model = next.model;
573
+ opts.onProgress?.(` 🧠 re-routed to an agentic-capable model: ${provider.name}${model ? ` / ${model}` : ''}`);
574
+ }
575
+ else {
576
+ return {
577
+ content: gate.notice ??
578
+ 'No agentic-capable model is available for this software task right now.',
579
+ followups: [],
580
+ generationFailed: true,
581
+ refused: true,
582
+ provider: type,
583
+ model,
584
+ transport: 'none',
585
+ };
586
+ }
587
+ }
588
+ else if (gate.notice) {
589
+ // proceed-weak-consented — say so plainly; never a silent weak model.
590
+ opts.onProgress?.(` ${gate.notice}`);
591
+ }
592
+ }
455
593
  // P4 — when a project is attached, recall its prior sessions + facts
456
594
  // FRESH per turn (the snapshot is cached, the recall is not — prior work
457
595
  // may have landed since the last turn). Best-effort: empty recall injects
@@ -513,11 +651,48 @@ export class ChatCommand extends BaseCommand {
513
651
  ...this.turnEnvelopeOf(answer),
514
652
  };
515
653
  }
654
+ // E1 — assemble the turn report from RECORDED evidence (plan store, tool
655
+ // outcomes, honesty flags). Derived, never narrated, so the trust verdict
656
+ // cannot be talked up by the model.
657
+ let turnReport;
658
+ try {
659
+ const planSnapshot = (opts.planStore ?? this.planStore).snapshot?.() ?? null;
660
+ turnReport = buildTurnReport({
661
+ goal: message,
662
+ plan: planSnapshot,
663
+ toolCalls: answer.toolCalls,
664
+ successfulToolCalls: answer.successfulToolCalls,
665
+ mutations: answer.runTrace?.mutations,
666
+ changedPaths: answer.runTrace?.paths,
667
+ flags: {
668
+ unverifiedActionClaim: answer.unverifiedActionClaim,
669
+ unverifiedEdit: answer.unverifiedEdit,
670
+ unverifiedEditClaim: answer.unverifiedEditClaim,
671
+ unverifiedBuildClaim: answer.unverifiedBuildClaim,
672
+ undeliveredArtifact: answer.undeliveredArtifact,
673
+ unfulfilledPromise: answer.unfulfilledPromise,
674
+ noActionTaken: answer.noActionTaken,
675
+ },
676
+ });
677
+ }
678
+ catch {
679
+ // A report must never break the turn.
680
+ }
681
+ // E-trace — persist the report on the turn's reasoning trace so it is
682
+ // reviewable after the fact (the Trace tab renders it), not only in this
683
+ // turn's return value. Best-effort: a trace write never breaks a turn.
684
+ recordTurnReport(answer.traceId, turnReport);
685
+ // E3 — surface a non-trivial report on the console. A plain answer (no
686
+ // plan, nothing changed) produces no summary and stays silent.
687
+ if (turnReport?.summary) {
688
+ opts.onProgress?.(formatTurnReport(turnReport));
689
+ }
516
690
  // E3b: strip raw suggest_followups JSON embedded in content by the model
517
691
  const cleanContent = stripToolCallArtifacts(answer.content || '');
518
692
  return {
519
693
  content: cleanContent,
520
694
  followups: answer.followups ?? [],
695
+ ...(turnReport ? { turnReport } : {}),
521
696
  generationFailed: answer.generationFailed,
522
697
  cancelled: answer.cancelled,
523
698
  bounded: answer.bounded,
@@ -812,7 +987,12 @@ export class ChatCommand extends BaseCommand {
812
987
  // so context-fit routing reacts to a long session, not just the task
813
988
  // text. Estimation only — never a hard block.
814
989
  const historyEstimate = estimateTokens(history.map((h) => h.content).join('\n') + '\n' + message);
815
- const routed = await this.routeMessageAuto(message, [], { contextHintTokens: historyEstimate });
990
+ // Intent-aware escalation for a bare continuation of software work.
991
+ const routingText = continuationSoftwareText(message, history) ?? undefined;
992
+ const routed = await this.routeMessageAuto(message, [], {
993
+ contextHintTokens: historyEstimate,
994
+ ...(routingText ? { routingText } : {}),
995
+ });
816
996
  type = routed.type;
817
997
  provider = routed.provider;
818
998
  effectiveModel = routed.model;
@@ -974,25 +1154,35 @@ export class ChatCommand extends BaseCommand {
974
1154
  const cacheModel = this.cacheModelFor(session);
975
1155
  if (cacheEnabled) {
976
1156
  try {
977
- const cachedResult = await cache.get(message, cacheModel, session.type, turnScope);
978
- if (cachedResult) {
1157
+ // #30 — read the entry WITH its recorded activity, not just the text: a
1158
+ // replay that dropped `toolCalls` rendered no tool cards on the dashboard
1159
+ // while the first run did, so a repeated prompt read as a turn that did
1160
+ // nothing. The text alone is still what the answer is; the activity is
1161
+ // reported so the surface is honest about what the cached turn DID.
1162
+ const cached = await cache.getEntry(message, cacheModel, session.type, turnScope);
1163
+ if (cached) {
979
1164
  // NOTE: the cached answer is NOT printed here — the caller prints
980
1165
  // content AFTER runChatAnswer returns (answer-first ordering). A
981
1166
  // print here would show the answer before the turn's own progress
982
1167
  // lines AND double-print it.
983
1168
  history.push({ role: 'user', content: message });
984
- history.push({ role: 'assistant', content: cachedResult });
985
- this.memoryNoteTurn(message, cachedResult);
1169
+ history.push({ role: 'assistant', content: cached.response });
1170
+ this.memoryNoteTurn(message, cached.response);
986
1171
  // WS2 — a cache replay reached no model, so the log says exactly that
987
1172
  // rather than borrowing an attribution from a turn that did not run.
988
1173
  // The workspace the replayed answer belongs to is recorded with the hit:
989
1174
  // a cache replay does no work, so "which project is this answer about?"
990
1175
  // is the one fact needed to tell a replay from a real turn.
991
- debugLog?.event('cache.hit', { chars: cachedResult.length, scope: turnScope });
1176
+ debugLog?.event('cache.hit', { chars: cached.response.length, scope: turnScope });
992
1177
  const cacheNotice = debugLogNotice(ctxOverrides?.debugSurface ?? 'cli-chat', debugLog?.write() ?? null);
993
1178
  if (cacheNotice)
994
1179
  logger.info(cacheNotice);
995
- return { content: cachedResult };
1180
+ return {
1181
+ content: cached.response,
1182
+ ...(cached.toolCalls ? { toolCalls: cached.toolCalls } : {}),
1183
+ ...(cached.successfulToolCalls ? { successfulToolCalls: cached.successfulToolCalls } : {}),
1184
+ ...(cached.bounded ? { bounded: cached.bounded } : {}),
1185
+ };
996
1186
  }
997
1187
  }
998
1188
  catch {
@@ -1112,17 +1302,23 @@ export class ChatCommand extends BaseCommand {
1112
1302
  // computed for the generation-FAILED fallback only — it does NOT reach the
1113
1303
  // prompt, so on this surface the MODEL decides and the rules are invisible.
1114
1304
  const systemText = buildToolSystemPrompt(parsed);
1115
- // Model-selected skills: a bounded CATALOG (name + one line) is appended to
1116
- // the system prompt and the MODEL decides which skill applies, loading it
1117
- // with the `skill` tool. This replaces keyword auto-injection, whose word
1118
- // lists could not tell "blood test report" from software testing — the model
1119
- // reads the same list and judges instantly, so there is no false positive to
1120
- // maintain away and no real match to accidentally drop. Best-effort: any
1121
- // failure returns '' and the turn proceeds byte-identically.
1305
+ // Skill hint — MODE-DEPENDENT (see resolveSkillHintMode):
1306
+ // - `match` (default) — the small keyword-matched hint: ONE skill, and
1307
+ // only when the goal really matches; otherwise nothing. This is the
1308
+ // 3.3.10 behaviour and keeps the prompt small.
1309
+ // - `catalog` (opt-in) — the full name+description catalog, which lets the
1310
+ // MODEL pick a skill but costs ~24K chars on every turn, so it must be
1311
+ // chosen (`NUVIRA_SKILL_CATALOG=catalog` or `skills.catalogHint`).
1312
+ // - `off` — never inject one.
1313
+ // Best-effort: any failure returns '' and the turn proceeds byte-identically.
1122
1314
  let skillHint = '';
1123
1315
  try {
1124
- const { buildSkillCatalogHint } = await import('../tools/loop-skill-hint.js');
1125
- skillHint = await buildSkillCatalogHint(this.configManager);
1316
+ const { buildConfiguredSkillHint, markLoopSkillUsed } = await import('../tools/loop-skill-hint.js');
1317
+ const injected = {
1318
+ value: null,
1319
+ };
1320
+ skillHint = await buildConfiguredSkillHint(message, this.configManager, injected);
1321
+ await markLoopSkillUsed(injected.value);
1126
1322
  }
1127
1323
  catch {
1128
1324
  skillHint = ''; // best-effort — a hint failure never breaks the turn
@@ -1281,8 +1477,23 @@ export class ChatCommand extends BaseCommand {
1281
1477
  }
1282
1478
  }
1283
1479
  // P0.7 — forward plan mutations to the GUI (structured checklist).
1284
- if (ctxOverrides?.onPlanChange && event === 'plan:changed') {
1285
- ctxOverrides.onPlanChange(data);
1480
+ if (event === 'plan:changed') {
1481
+ const snapshot = data;
1482
+ if (ctxOverrides?.onPlanChange) {
1483
+ ctxOverrides.onPlanChange(snapshot);
1484
+ }
1485
+ else {
1486
+ // No GUI consumer (the interactive CLI): show the PROGRESS TABLE in
1487
+ // the terminal so a user watching the run sees the plan advance,
1488
+ // not just the model's narration. A settled plan prints its
1489
+ // achieved SUMMARY instead — the same text the turn closes on.
1490
+ const store = ctxOverrides?.planStore ?? this.planStore;
1491
+ const settled = snapshot.steps.length > 0 &&
1492
+ snapshot.steps.every((s) => s.status === 'done' || s.status === 'blocked');
1493
+ const rendered = settled ? store.summary?.() : store.toTable?.();
1494
+ if (rendered)
1495
+ logger.info(`\n${rendered}\n`);
1496
+ }
1286
1497
  }
1287
1498
  // P3b — forward git diff payloads to the GUI (the diff card).
1288
1499
  if (ctxOverrides?.onGitDiff && event === 'git:diff') {
@@ -1332,6 +1543,13 @@ export class ChatCommand extends BaseCommand {
1332
1543
  // default interactive renderer.
1333
1544
  ...(ctxOverrides?.askUser ? { askUser: ctxOverrides.askUser } : {}),
1334
1545
  ...(ctxOverrides?.gateway ? { gateway: ctxOverrides.gateway } : {}),
1546
+ // P0.7 — the plan store. This was DROPPED here: `answerOnce` put a
1547
+ // per-session store on `ctxOverrides.planStore`, but only askUser/gateway
1548
+ // were threaded into the tool context, so every surface fell back to the
1549
+ // shared module store — which is why plans leaked across sessions and a
1550
+ // reload started a blank checklist. Thread it (per-session when injected,
1551
+ // else this command's project-scoped store).
1552
+ planStore: ctxOverrides?.planStore ?? this.planStore,
1335
1553
  // C2 verify with the actual session model (verify_requirement tool).
1336
1554
  callLLM: (prompt, opts) => session.provider.generate(prompt, { ...opts, model: session.model }),
1337
1555
  // I3: tools that return {artifact, result} deliverables are recorded to
@@ -1349,7 +1567,7 @@ export class ChatCommand extends BaseCommand {
1349
1567
  },
1350
1568
  },
1351
1569
  };
1352
- const callModel = this.buildToolCallModel(message, session, options, mode, ctxOverrides?.onToken, ctxOverrides?.signal);
1570
+ const callModel = this.buildToolCallModel(message, session, options, mode, ctxOverrides?.onToken, ctxOverrides?.signal, continuationSoftwareText(message, history) ?? undefined);
1353
1571
  let result;
1354
1572
  // v1.8x audit — CHAT TRACE CAPTURE: every LLM call in a chat turn is now
1355
1573
  // recorded to ~/.nuvira/memory/reasoning-traces.json (source 'chat'), so
@@ -1367,6 +1585,64 @@ export class ChatCommand extends BaseCommand {
1367
1585
  });
1368
1586
  // G18 — the tool-context emit (declared above) now has somewhere to write.
1369
1587
  traceIdForEvents = chatTraceId;
1588
+ // A2 — record the routing DECISION as a first-class event, so a turn that
1589
+ // ran on a weak/incapable model is self-evident in the Trace tab. The
1590
+ // failed Tauri turn had no such record; this is the instrument that would
1591
+ // have shown `agenticCapable:false` at the moment of the choice.
1592
+ if (this.lastRouteSnapshot) {
1593
+ const snap = this.lastRouteSnapshot;
1594
+ recordTraceEvent(chatTraceId, {
1595
+ kind: 'decision',
1596
+ gate: 'routing',
1597
+ summary: `routed to ${snap.provider}/${snap.model} ` +
1598
+ `(complexity ${snap.complexity}, score ${snap.score.toFixed(3)})` +
1599
+ (snap.agenticCapable === false
1600
+ ? ` — NOT agentic-capable${snap.overrideReason ? ` (${snap.overrideReason})` : ''}`
1601
+ : ''),
1602
+ routing: snap,
1603
+ });
1604
+ }
1605
+ // D1 — measure the OUTBOUND context and, past the ceiling, degrade the
1606
+ // lowest-value optional contributor first (skill hint → work digest →
1607
+ // recall → …). The identity/tool contract is never trimmed. This is the
1608
+ // guard the 3.3.11 bloat (7.6K → 32.8K chars) never had.
1609
+ try {
1610
+ const contextBudget = resolvePromptBudget(this.configManager);
1611
+ const historyChars = history.reduce((n, h) => n + (h.content?.length ?? 0), 0);
1612
+ const report = measurePromptBudget([
1613
+ { name: 'system:identity+tool-contract', chars: systemText.length },
1614
+ { name: 'system:channel-policy', chars: systemPolicyBlock.length },
1615
+ { name: 'skill-hint', chars: skillHint.length, dropPriority: 10 },
1616
+ { name: 'working-state', chars: workingStateBlock.length, dropPriority: 25 },
1617
+ { name: 'recall', chars: ctxOverrides?.recallContext?.length ?? 0, dropPriority: 30 },
1618
+ {
1619
+ name: 'project-context',
1620
+ chars: ctxOverrides?.projectContext?.length ?? ambientProjectContext?.length ?? 0,
1621
+ dropPriority: 40,
1622
+ },
1623
+ { name: 'file-context', chars: fileContext?.length ?? 0, dropPriority: 45 },
1624
+ { name: 'history+ask', chars: historyChars },
1625
+ ], { budget: contextBudget });
1626
+ // Apply the FIRST ladder step (the documented, safe one): drop the skill
1627
+ // hint from the assembled system message.
1628
+ if (report.trims.includes('skill-hint') && skillHint) {
1629
+ thread[0] = { role: 'system', content: systemText + systemPolicyBlock };
1630
+ }
1631
+ if (report.level !== 'ok') {
1632
+ recordTraceEvent(chatTraceId, {
1633
+ kind: 'decision',
1634
+ gate: 'context-budget',
1635
+ summary: formatPromptBudgetBreakdown(report) +
1636
+ (report.trims.length ? ` — trimmed: ${report.trims.join(', ')}` : ''),
1637
+ });
1638
+ }
1639
+ }
1640
+ catch {
1641
+ // A budget measurement must never break the turn.
1642
+ }
1643
+ // The instant this turn's model walk began. A failed generation records the
1644
+ // WALK (below) relative to this mark, so the trace can say what was tried.
1645
+ const modelWalkMark = Date.now();
1370
1646
  const seenStepDigests = new Set();
1371
1647
  const digestPrompt = (p) => {
1372
1648
  try {
@@ -1498,6 +1774,36 @@ export class ChatCommand extends BaseCommand {
1498
1774
  // (`answerQualityError`), which the loop rethrows once every candidate has
1499
1775
  // narrated.
1500
1776
  logger.error(String(err));
1777
+ // DIAGNOSABILITY — a failed turn used to record only the LAST attempt's
1778
+ // provider/model beside the FIRST error, so a reader could not tell which
1779
+ // model actually ran, nor WHY other models were not used. Record the WALK:
1780
+ // what was tried, what was parked (and for how long), and how large the
1781
+ // eligible pool was — the facts that separate a real shortage from a
1782
+ // routing gap. This is the exact ambiguity in the traces that prompted it:
1783
+ // a step labelled `local/qwen2.5:0.5b` carrying Gemini's 429.
1784
+ try {
1785
+ const report = modelBreadthReport(modelWalkMark, this.configManager);
1786
+ const tried = report.tried.filter((a) => !a.skipped);
1787
+ const parked = report.parked.filter((r) => r.active);
1788
+ const triedList = tried
1789
+ .slice(0, 6)
1790
+ .map((a) => `${a.provider}/${a.model} (${a.reason})`)
1791
+ .join(', ');
1792
+ const parkedList = parked
1793
+ .slice(0, 6)
1794
+ .map((r) => `${r.provider}${r.model ? `/${r.model}` : ''} (${r.kind})`)
1795
+ .join(', ');
1796
+ recordTraceEvent(chatTraceId, {
1797
+ kind: 'failover',
1798
+ summary: `the model layer failed — eligible pool ${report.poolSize ?? '?'} model(s) across ` +
1799
+ `${report.poolProviders ?? '?'} provider(s), ${tried.length} tried, ${parked.length} parked` +
1800
+ (triedList ? `; tried: ${triedList}` : '') +
1801
+ (parkedList ? `; parked: ${parkedList}` : ''),
1802
+ });
1803
+ }
1804
+ catch {
1805
+ // Diagnosis is a courtesy — it must never break the failure path.
1806
+ }
1501
1807
  endTrace(chatTraceId, false, { kind: 'failed' });
1502
1808
  result = {
1503
1809
  // Sanitized on purpose: this content is delivered verbatim by every
@@ -1609,11 +1915,29 @@ export class ChatCommand extends BaseCommand {
1609
1915
  if (result.content.trim() && !result.generationFailed && !result.cancelled) {
1610
1916
  if (cacheEnabled) {
1611
1917
  try {
1612
- // Keyed by the model that ACTUALLY answered (tryGenerate records it
1613
- // on success), so a weak model's reply is never replayed as a strong
1614
- // model's. `cacheModel` is the pre-flight fallback for the paths that
1615
- // never resolve one (e.g. a cached-hit turn).
1616
- await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type, undefined, turnScope);
1918
+ // #30 — NEVER cache a turn that carries an honesty flag. The cache
1919
+ // stores text only, so storing a flagged answer would let a later
1920
+ // identical prompt (or the same prompt on another surface) replay a
1921
+ // known-unverified claim as a clean one — a truthfulness hole the whole
1922
+ // flag system exists to close. A flagged turn re-derives instead.
1923
+ if (turnCarriesHonestyFlag(result)) {
1924
+ debugLog?.event('cache.skip', { reason: 'honesty-flag', scope: turnScope });
1925
+ }
1926
+ else {
1927
+ // Keyed by the model that ACTUALLY answered (tryGenerate records it
1928
+ // on success), so a weak model's reply is never replayed as a strong
1929
+ // model's. `cacheModel` is the pre-flight fallback for the paths that
1930
+ // never resolve one (e.g. a cached-hit turn).
1931
+ //
1932
+ // #30 — the turn ACTIVITY rides with the text so a replay reports what
1933
+ // the cached turn did (tool cards, bounded), rather than reading as a
1934
+ // turn that did nothing.
1935
+ await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type, undefined, turnScope, {
1936
+ ...(result.toolCalls ? { toolCalls: result.toolCalls } : {}),
1937
+ ...(result.successfulToolCalls ? { successfulToolCalls: result.successfulToolCalls } : {}),
1938
+ ...(result.bounded ? { bounded: true } : {}),
1939
+ });
1940
+ }
1617
1941
  }
1618
1942
  catch {
1619
1943
  // Best-effort.
@@ -1710,6 +2034,15 @@ export class ChatCommand extends BaseCommand {
1710
2034
  transport: result.transport,
1711
2035
  // WS1 — the findings this turn recorded, with their verdicts.
1712
2036
  findings,
2037
+ // E — the honest "what happened" facts the TurnReport is derived from.
2038
+ successfulToolCalls: result.successfulToolCalls,
2039
+ runTrace: result.runTrace,
2040
+ unverifiedEdit: result.unverifiedEdit,
2041
+ unverifiedEditClaim: result.unverifiedEditClaim,
2042
+ noActionTaken: result.noActionTaken,
2043
+ // E-trace — the trace this turn was recorded under, so `answerOnce` can
2044
+ // attach the TurnReport to it (see `recordTurnReport`).
2045
+ traceId: chatTraceId,
1713
2046
  });
1714
2047
  }
1715
2048
  /**
@@ -1743,7 +2076,15 @@ export class ChatCommand extends BaseCommand {
1743
2076
  return `${session.type}:unresolved`;
1744
2077
  }
1745
2078
  }
1746
- buildToolCallModel(message, session, options, mode, onToken, signal) {
2079
+ buildToolCallModel(message, session, options, mode, onToken, signal,
2080
+ /**
2081
+ * Intent-aware failover: the prior software ask when this turn is a bare
2082
+ * continuation ("yes", "do it"), computed by the caller from the history
2083
+ * it owns. The initial route used it; the mid-turn failover walk needs it
2084
+ * too, or a continuation whose first cloud provider dies re-routes on the
2085
+ * signal-free message and can land on a tiny local model.
2086
+ */
2087
+ routingText) {
1747
2088
  return async (messages, schemas, stepOnToken, stepSignal) => {
1748
2089
  // The effective token sink: the caller's stream wins; when a step-level
1749
2090
  // sink is also given (loop passthrough) they are the same channel.
@@ -1953,13 +2294,27 @@ export class ChatCommand extends BaseCommand {
1953
2294
  // success log per landed candidate.
1954
2295
  logger.warn(` ⚠️ ${session.provider.name} failed — trying the next auto candidate...`);
1955
2296
  const failed = new Set([session.type]);
2297
+ // Intent-aware escalation for the FAILOVER walk (the live junk path):
2298
+ // a mid-turn failure used to re-route on the bare message alone, so a
2299
+ // software turn whose first cloud provider died fell through to a tiny
2300
+ // local model (`local/qwen2.5:0.5b`) that FABRICATED tool output
2301
+ // (trace-1791118650644-d73hyr). Pass the same prior-software-ask hint
2302
+ // the initial route used, so the router's agentic capability floor
2303
+ // applies here too — but NEVER drop a candidate the floor would keep,
2304
+ // so auto still cannot dead-end (local stays the last resort).
2305
+ const failoverRoutingText = routingText;
1956
2306
  // Try ALL ranked candidates (no 3-candidate cap) — bounded by
1957
2307
  // the number of known providers to prevent infinite loops.
1958
2308
  const maxAttempts = 10;
1959
2309
  for (let i = 0; i < maxAttempts; i++) {
1960
2310
  let next = null;
1961
2311
  try {
1962
- next = await this.routeMessageAuto(message, [...failed]);
2312
+ next = failoverRoutingText
2313
+ ? await this.routeMessageAuto(message, [...failed], {
2314
+ routingText: failoverRoutingText,
2315
+ fallbackFrom: firstType,
2316
+ })
2317
+ : await this.routeMessageAuto(message, [...failed], { fallbackFrom: firstType });
1963
2318
  }
1964
2319
  catch {
1965
2320
  break;
@@ -1967,6 +2322,17 @@ export class ChatCommand extends BaseCommand {
1967
2322
  if (!next || next.type === session.type || failed.has(next.type))
1968
2323
  break;
1969
2324
  failed.add(next.type);
2325
+ // Last-resort truthfulness: if even the agentic floor could not
2326
+ // avoid a weak model, say so ONCE so a degraded answer is never
2327
+ // mistaken for a real one (the tiny local model fabricated tool
2328
+ // results instead of admitting it could not run them). C2 — the
2329
+ // wording comes from the SHARED helper so chat, the orchestrator
2330
+ // and the dashboard cannot describe this three different ways.
2331
+ if (!isAgenticCapableModel(next.model, next.type)) {
2332
+ const notice = weakRouteNotice({ provider: next.type, model: next.model }, true) ??
2333
+ `⚠️ no agentic-capable model left — falling back to ${next.type}/${next.model}`;
2334
+ logger.warn(` no agentic-capable model left — ${notice}`);
2335
+ }
1970
2336
  // Opt-in confirmation (routing.promptOnFailover): 'manual' stops
1971
2337
  // the walk and lets the caller's error recovery handle it. Gated on
1972
2338
  // an interactive stdin (inherited from the shared single-shot
@@ -2218,6 +2584,10 @@ export class ChatCommand extends BaseCommand {
2218
2584
  * so Auto routing never sends a request to a provider that would 401.
2219
2585
  */
2220
2586
  async routeMessageAuto(message, excludeProviders = [], opts) {
2587
+ // The text ROUTING is decided from — the prior software ask for a bare
2588
+ // continuation, otherwise the message itself. `message` is used everywhere
2589
+ // else (the answer, the tool loop).
2590
+ const taskText = opts?.routingText?.trim() ? opts.routingText : message;
2221
2591
  // Feed the SHARED circuit breaker into the router so a provider that has
2222
2592
  // failed repeatedly (recorded by recordFailure below) is deprioritized by
2223
2593
  // scoring, not just skipped by the candidate walk.
@@ -2235,7 +2605,7 @@ export class ChatCommand extends BaseCommand {
2235
2605
  // circuit-breaker state on top.
2236
2606
  // C3: the NLU parser seeds the router task-intent (same vocabulary every
2237
2607
  // action command derives from resolveDispatch) when confident.
2238
- const parsed = parseRequestSync(message);
2608
+ const parsed = parseRequestSync(taskText);
2239
2609
  const dispatch = resolveDispatch(parsed);
2240
2610
  // Routing decision cache (assessment v4 Phase 2): the loop engine
2241
2611
  // resolves per turn; turns with identical STABLE routing inputs (intent,
@@ -2259,7 +2629,7 @@ export class ChatCommand extends BaseCommand {
2259
2629
  const cacheSignature = routingCacheSignature([
2260
2630
  'chat',
2261
2631
  dispatch.taskIntentHint ?? null,
2262
- analyzeComplexity(message),
2632
+ analyzeComplexity(taskText),
2263
2633
  routingCfg.preferenceMode ?? null,
2264
2634
  routingCfg.bandit === false ? 'b' : 'B',
2265
2635
  routingCfg.mlRouter === true ? 'm' : 'M',
@@ -2272,7 +2642,7 @@ export class ChatCommand extends BaseCommand {
2272
2642
  .sort()
2273
2643
  .join(','),
2274
2644
  ]);
2275
- const decision = withRoutingCache(cacheSignature, 30_000, () => getAutoRouter().resolve('chat', message, {
2645
+ const decision = withRoutingCache(cacheSignature, 30_000, () => getAutoRouter().resolve('chat', taskText, {
2276
2646
  ...buildAutoResolveOptions(this.configManager, {
2277
2647
  verbose: envBuff('DEBUG') === 'true',
2278
2648
  contextHintTokens: opts?.contextHintTokens,
@@ -2288,6 +2658,10 @@ export class ChatCommand extends BaseCommand {
2288
2658
  score: decision.score,
2289
2659
  complexity: String(decision.complexity),
2290
2660
  explanation: decision.explanation,
2661
+ // A1/A2/C3 — carry the capability verdict + override reason so the trace
2662
+ // and console can show WHY the pair was chosen, not just which pair.
2663
+ agenticCapable: decision.agenticCapable,
2664
+ overrideReason: decision.overrideReason,
2291
2665
  };
2292
2666
  // Walk the ranked candidates (winner first) and return the first available
2293
2667
  // provider — never a provider that lacks a key or endpoint. Providers that
@@ -2446,6 +2820,9 @@ export class ChatCommand extends BaseCommand {
2446
2820
  provider: candidate.provider,
2447
2821
  model,
2448
2822
  score: decision.score,
2823
+ agenticCapable: isAgenticCapableModel(model, candidate.provider),
2824
+ overrideReason: decision.overrideReason,
2825
+ ...(opts?.fallbackFrom ? { fallbackFrom: opts.fallbackFrom } : {}),
2449
2826
  });
2450
2827
  return {
2451
2828
  type: resolved.type,
@@ -2454,6 +2831,12 @@ export class ChatCommand extends BaseCommand {
2454
2831
  ranked: candidates,
2455
2832
  complexity: decision.complexity,
2456
2833
  score: decision.score,
2834
+ agenticCapable: isAgenticCapableModel(model, resolved.type),
2835
+ overrideReason: decision.overrideReason,
2836
+ taskProfile: {
2837
+ intent: decision.taskProfile.intent,
2838
+ requiresVerification: decision.taskProfile.requiresVerification,
2839
+ },
2457
2840
  };
2458
2841
  }
2459
2842
  }
@@ -2477,6 +2860,9 @@ export class ChatCommand extends BaseCommand {
2477
2860
  provider: usableProvider,
2478
2861
  model: decision.model,
2479
2862
  score: decision.score,
2863
+ agenticCapable: isAgenticCapableModel(decision.model, usableProvider),
2864
+ overrideReason: decision.overrideReason,
2865
+ ...(opts?.fallbackFrom ? { fallbackFrom: opts.fallbackFrom } : {}),
2480
2866
  });
2481
2867
  const resolved = resolveProvider(this.configManager, usableProvider);
2482
2868
  const model = (await resolveRoute({