agent-nuvira 3.3.11 → 3.3.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -1
- package/dist/agents/agents/reasoner.d.ts.map +1 -1
- package/dist/agents/agents/reasoner.js +13 -0
- package/dist/agents/agents/reasoner.js.map +1 -1
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +56 -20
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/cli/chat.d.ts +36 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +456 -43
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts.map +1 -1
- package/dist/cli/cli-program.js +8 -0
- package/dist/cli/cli-program.js.map +1 -1
- package/dist/cli/config.d.ts.map +1 -1
- package/dist/cli/config.js +35 -4
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +14 -0
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/knowledge.d.ts +23 -0
- package/dist/cli/knowledge.d.ts.map +1 -0
- package/dist/cli/knowledge.js +132 -0
- package/dist/cli/knowledge.js.map +1 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +57 -13
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/omniroute-control.d.ts +88 -0
- package/dist/cli/omniroute-control.d.ts.map +1 -0
- package/dist/cli/omniroute-control.js +179 -0
- package/dist/cli/omniroute-control.js.map +1 -0
- package/dist/cli/omniroute.d.ts +21 -0
- package/dist/cli/omniroute.d.ts.map +1 -0
- package/dist/cli/omniroute.js +67 -0
- package/dist/cli/omniroute.js.map +1 -0
- package/dist/cli/tool-install-prompt.d.ts +58 -0
- package/dist/cli/tool-install-prompt.d.ts.map +1 -1
- package/dist/cli/tool-install-prompt.js +137 -0
- package/dist/cli/tool-install-prompt.js.map +1 -1
- package/dist/cli/weak-model-prompt.d.ts +16 -7
- package/dist/cli/weak-model-prompt.d.ts.map +1 -1
- package/dist/cli/weak-model-prompt.js +24 -8
- package/dist/cli/weak-model-prompt.js.map +1 -1
- package/dist/config/types.d.ts +44 -0
- package/dist/config/types.d.ts.map +1 -1
- package/dist/context/cache.d.ts +35 -1
- package/dist/context/cache.d.ts.map +1 -1
- package/dist/context/cache.js +31 -2
- package/dist/context/cache.js.map +1 -1
- package/dist/context/history.d.ts +21 -0
- package/dist/context/history.d.ts.map +1 -1
- package/dist/context/history.js +50 -0
- package/dist/context/history.js.map +1 -1
- package/dist/gateway/hook-contract.d.ts +173 -0
- package/dist/gateway/hook-contract.d.ts.map +1 -0
- package/dist/gateway/hook-contract.js +352 -0
- package/dist/gateway/hook-contract.js.map +1 -0
- package/dist/gateway/hooks.d.ts.map +1 -1
- package/dist/gateway/hooks.js +14 -0
- package/dist/gateway/hooks.js.map +1 -1
- package/dist/inference/interface.d.ts +17 -0
- package/dist/inference/interface.d.ts.map +1 -1
- package/dist/inference/model-catalog.d.ts +17 -0
- package/dist/inference/model-catalog.d.ts.map +1 -1
- package/dist/inference/model-catalog.js +42 -7
- package/dist/inference/model-catalog.js.map +1 -1
- package/dist/inference/model-id-validation.d.ts +57 -0
- package/dist/inference/model-id-validation.d.ts.map +1 -0
- package/dist/inference/model-id-validation.js +127 -0
- package/dist/inference/model-id-validation.js.map +1 -0
- package/dist/inference/openai-compat-adapter.d.ts.map +1 -1
- package/dist/inference/openai-compat-adapter.js +12 -1
- package/dist/inference/openai-compat-adapter.js.map +1 -1
- package/dist/inference/provider-catalog.d.ts +11 -0
- package/dist/inference/provider-catalog.d.ts.map +1 -1
- package/dist/inference/provider-catalog.js +80 -0
- package/dist/inference/provider-catalog.js.map +1 -1
- package/dist/inference/route-resolver.d.ts +2 -8
- package/dist/inference/route-resolver.d.ts.map +1 -1
- package/dist/inference/route-resolver.js +17 -0
- package/dist/inference/route-resolver.js.map +1 -1
- package/dist/inference/tool-call-utils.d.ts +4 -3
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +26 -12
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/inference/tools.d.ts +1 -0
- package/dist/inference/tools.d.ts.map +1 -1
- package/dist/inference/tools.js +20 -3
- package/dist/inference/tools.js.map +1 -1
- package/dist/learning/agentic-route-gate.d.ts +104 -0
- package/dist/learning/agentic-route-gate.d.ts.map +1 -0
- package/dist/learning/agentic-route-gate.js +125 -0
- package/dist/learning/agentic-route-gate.js.map +1 -0
- package/dist/learning/auto-router.d.ts +89 -0
- package/dist/learning/auto-router.d.ts.map +1 -1
- package/dist/learning/auto-router.js +188 -8
- package/dist/learning/auto-router.js.map +1 -1
- package/dist/learning/build-prerequisites.d.ts +68 -0
- package/dist/learning/build-prerequisites.d.ts.map +1 -0
- package/dist/learning/build-prerequisites.js +267 -0
- package/dist/learning/build-prerequisites.js.map +1 -0
- package/dist/learning/continuation-intent.d.ts +43 -0
- package/dist/learning/continuation-intent.d.ts.map +1 -0
- package/dist/learning/continuation-intent.js +82 -0
- package/dist/learning/continuation-intent.js.map +1 -0
- package/dist/learning/error-repair.d.ts +23 -1
- package/dist/learning/error-repair.d.ts.map +1 -1
- package/dist/learning/error-repair.js +58 -0
- package/dist/learning/error-repair.js.map +1 -1
- package/dist/learning/knowledge-base.d.ts +124 -0
- package/dist/learning/knowledge-base.d.ts.map +1 -0
- package/dist/learning/knowledge-base.js +337 -0
- package/dist/learning/knowledge-base.js.map +1 -0
- package/dist/learning/model-harness.d.ts +19 -0
- package/dist/learning/model-harness.d.ts.map +1 -1
- package/dist/learning/model-harness.js +27 -0
- package/dist/learning/model-harness.js.map +1 -1
- package/dist/learning/model-selection.d.ts +2 -1
- package/dist/learning/model-selection.d.ts.map +1 -1
- package/dist/learning/model-selection.js +35 -2
- package/dist/learning/model-selection.js.map +1 -1
- package/dist/learning/project-docs.d.ts +83 -0
- package/dist/learning/project-docs.d.ts.map +1 -0
- package/dist/learning/project-docs.js +132 -0
- package/dist/learning/project-docs.js.map +1 -0
- package/dist/learning/prompt-budget.d.ts +91 -0
- package/dist/learning/prompt-budget.d.ts.map +1 -0
- package/dist/learning/prompt-budget.js +99 -0
- package/dist/learning/prompt-budget.js.map +1 -0
- package/dist/learning/quota-ledger.d.ts +13 -0
- package/dist/learning/quota-ledger.d.ts.map +1 -1
- package/dist/learning/quota-ledger.js +28 -0
- package/dist/learning/quota-ledger.js.map +1 -1
- package/dist/learning/reasoning-trace.d.ts +55 -2
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +48 -0
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +16 -3
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/routing-history.d.ts +13 -0
- package/dist/learning/routing-history.d.ts.map +1 -1
- package/dist/learning/routing-history.js.map +1 -1
- package/dist/learning/seeded-benchmark.d.ts +14 -0
- package/dist/learning/seeded-benchmark.d.ts.map +1 -1
- package/dist/learning/seeded-benchmark.js +43 -4
- package/dist/learning/seeded-benchmark.js.map +1 -1
- package/dist/learning/turn-report.d.ts +78 -0
- package/dist/learning/turn-report.d.ts.map +1 -0
- package/dist/learning/turn-report.js +140 -0
- package/dist/learning/turn-report.js.map +1 -0
- package/dist/security/secret-scan.d.ts +156 -0
- package/dist/security/secret-scan.d.ts.map +1 -0
- package/dist/security/secret-scan.js +525 -0
- package/dist/security/secret-scan.js.map +1 -0
- package/dist/tools/edit-verification.d.ts +2 -11
- package/dist/tools/edit-verification.d.ts.map +1 -1
- package/dist/tools/edit-verification.js +33 -1
- package/dist/tools/edit-verification.js.map +1 -1
- package/dist/tools/loop-skill-hint.d.ts +51 -0
- package/dist/tools/loop-skill-hint.d.ts.map +1 -1
- package/dist/tools/loop-skill-hint.js +87 -2
- package/dist/tools/loop-skill-hint.js.map +1 -1
- package/dist/tools/plan-store.d.ts +126 -3
- package/dist/tools/plan-store.d.ts.map +1 -1
- package/dist/tools/plan-store.js +290 -6
- package/dist/tools/plan-store.js.map +1 -1
- package/dist/tools/registry.d.ts +20 -1
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +165 -4
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/remediation-ladder.d.ts +91 -0
- package/dist/tools/remediation-ladder.d.ts.map +1 -0
- package/dist/tools/remediation-ladder.js +308 -0
- package/dist/tools/remediation-ladder.js.map +1 -0
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +60 -2
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/skills-hub.d.ts +7 -0
- package/dist/tools/skills-hub.d.ts.map +1 -1
- package/dist/tools/skills-hub.js +11 -2
- package/dist/tools/skills-hub.js.map +1 -1
- package/dist/tools/step-artifact.d.ts +35 -0
- package/dist/tools/step-artifact.d.ts.map +1 -0
- package/dist/tools/step-artifact.js +154 -0
- package/dist/tools/step-artifact.js.map +1 -0
- package/dist/tools/tool-loop.d.ts +35 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +284 -9
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/tools/toolsets.js +4 -4
- package/dist/tools/toolsets.js.map +1 -1
- package/dist/tools/worktree.d.ts +11 -2
- package/dist/tools/worktree.d.ts.map +1 -1
- package/dist/tools/worktree.js +11 -2
- package/dist/tools/worktree.js.map +1 -1
- package/dist/utils/effect-verification.js +1 -1
- package/dist/utils/effect-verification.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +21 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +33 -5
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/server.d.ts +29 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +410 -5
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +137 -0
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.d.ts +7 -0
- package/dist/web-dashboard/workspace-guard.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.js +70 -0
- package/dist/web-dashboard/workspace-guard.js.map +1 -1
- package/package.json +4 -1
- package/src/web-dashboard/public/assets/index-DhiuR4S2.js +210 -0
- package/src/web-dashboard/public/assets/index-DhiuR4S2.js.map +1 -0
- package/src/web-dashboard/public/assets/index-tK8Y2qHB.css +1 -0
- package/src/web-dashboard/public/index.html +2 -2
- package/src/web-dashboard/public/assets/index-BdKf5Xw2.js +0 -207
- package/src/web-dashboard/public/assets/index-BdKf5Xw2.js.map +0 -1
- package/src/web-dashboard/public/assets/index-C9eBskc9.css +0 -1
package/dist/cli/chat.js
CHANGED
|
@@ -5,7 +5,7 @@ import { Command } from 'commander';
|
|
|
5
5
|
import inquirer from 'inquirer';
|
|
6
6
|
import { BaseCommand } from './commands.js';
|
|
7
7
|
import { resolveProvider } from './router.js';
|
|
8
|
-
import { resolveRoute } from '../inference/route-resolver.js';
|
|
8
|
+
import { resolveRoute, strictModelMode } from '../inference/route-resolver.js';
|
|
9
9
|
import { showModelPicker } from './model-picker.js';
|
|
10
10
|
import { ContextParser } from '../context/parser.js';
|
|
11
11
|
import { getCache } from '../context/cache.js';
|
|
@@ -26,7 +26,8 @@ import { capabilityReasoningEffort, isMaxCapability } from '../config/capability
|
|
|
26
26
|
import { getProviderFallback, classifyFallbackError, isRetryableError, isTransientForRetry, recordRegistrySuccess } from '../learning/provider-fallback.js';
|
|
27
27
|
import { recordActionFailure } from '../learning/failure-bookkeeping.js';
|
|
28
28
|
import { resolveThreadBudgetChars } from '../learning/context-budget.js';
|
|
29
|
-
import { getAutoRouter, isAutoModel, isAutoProvider, governanceVerdict } from '../learning/auto-router.js';
|
|
29
|
+
import { getAutoRouter, isAgentAutoRoute, isAutoModel, isAutoProvider, governanceVerdict, adminBudgetVerdict } from '../learning/auto-router.js';
|
|
30
|
+
import { continuationSoftwareText } from '../learning/continuation-intent.js';
|
|
30
31
|
import { estimateTokens } from '../learning/cost-tracker.js';
|
|
31
32
|
import { getModelRegistry } from '../learning/model-registry.js';
|
|
32
33
|
import { refreshModelRegistry } from '../inference/model-probe.js';
|
|
@@ -34,9 +35,9 @@ import { startWarmupDaemon } from '../learning/model-warmup.js';
|
|
|
34
35
|
import { recordRoutingDecision } from '../learning/routing-history.js';
|
|
35
36
|
import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
|
|
36
37
|
import { buildAutoResolveOptions } from '../learning/resolve-options.js';
|
|
37
|
-
import { buildDeepFailoverPool, createFailoverExclusionFilter } from '../learning/resilient-call.js';
|
|
38
|
+
import { buildDeepFailoverPool, createFailoverExclusionFilter, modelBreadthReport } from '../learning/resilient-call.js';
|
|
38
39
|
import { parseRequestSync } from '../nlu/parser.js';
|
|
39
|
-
import {
|
|
40
|
+
import { createPersistentPlanStore } from '../tools/plan-store.js';
|
|
40
41
|
import { withLogCorrelation } from '../enterprise/log.js';
|
|
41
42
|
import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
|
|
42
43
|
import { resolveDispatch } from '../nlu/actions.js';
|
|
@@ -45,10 +46,13 @@ import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
|
|
|
45
46
|
// WS3 (#25) — the turn as a span, when an operator has asked for OTLP export.
|
|
46
47
|
import { flushSpans, otelNoticeOnce, startTurnSpan } from '../observability/otel.js';
|
|
47
48
|
import { detectAnswerQualityFailure, answerQualityError, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
|
|
48
|
-
import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, buildTraceOutcome, traceOutcomeSucceeded } from '../learning/reasoning-trace.js';
|
|
49
|
+
import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, recordTurnReport, buildTraceOutcome, traceOutcomeSucceeded } from '../learning/reasoning-trace.js';
|
|
49
50
|
import { recordWorkingState, getWorkingState, formatWorkingState, isProjectLedgerDir } from '../learning/working-state.js';
|
|
50
51
|
import { getLoopExposureMode } from '../tools/toolsets.js';
|
|
51
|
-
import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
|
|
52
|
+
import { resolveModelHarnessProfile, shouldSkipNativeTools, isAgenticCapableModel } from '../learning/model-harness.js';
|
|
53
|
+
import { assertAgenticRoute, setWeakModelConsent, resolveWeakModelPolicy, weakRouteNotice } from '../learning/agentic-route-gate.js';
|
|
54
|
+
import { resolvePromptBudget, measurePromptBudget, formatPromptBudgetBreakdown } from '../learning/prompt-budget.js';
|
|
55
|
+
import { buildTurnReport, formatTurnReport } from '../learning/turn-report.js';
|
|
52
56
|
import { resolveAdapterDefault, hasCredentials } from '../learning/model-selection.js';
|
|
53
57
|
import { buildLoopProjectContext } from '../tools/loop-project-context.js';
|
|
54
58
|
import { sweepTransientFailures, collectionRevivalStore } from '../learning/provider-revival.js';
|
|
@@ -281,7 +285,12 @@ export async function generateWithTransientRetry(attempt, signal, onRetry) {
|
|
|
281
285
|
}
|
|
282
286
|
}
|
|
283
287
|
}
|
|
284
|
-
|
|
288
|
+
/**
|
|
289
|
+
* Exported for the release gate (`tests/release/agent-contracts.test.ts`): the
|
|
290
|
+
* assembled system prompt's SIZE and its required contract clauses are a
|
|
291
|
+
* release invariant, not an implementation detail. See the gate for why.
|
|
292
|
+
*/
|
|
293
|
+
export function buildToolSystemPrompt(parsed) {
|
|
285
294
|
return [
|
|
286
295
|
"You are Nuvira, Agent-Nuvira's AI agent. You code, create, write, analyze, and automate — anything the user needs. You identify as Nuvira (never 'Buff').",
|
|
287
296
|
'Be precise and honest. When a request is ambiguous or incomplete, clarify with ask_user instead of guessing.',
|
|
@@ -316,9 +325,43 @@ function buildToolSystemPrompt(parsed) {
|
|
|
316
325
|
'- Never ask the user to attach a folder, and never tell them to move files into a directory. If a request genuinely needs a workspace and none is attached, the app asks for one on its own — so just answer as best you can.',
|
|
317
326
|
'- Treat a request as project work ONLY when it actually concerns the codebase or produces a file on disk. A general question that happens to contain a word like "class", "project", "test" or "document" is still a general question.',
|
|
318
327
|
'',
|
|
328
|
+
// WHY THIS BLOCK EXISTS. Asked to build a Tauri app, a model ran
|
|
329
|
+
// `cargo --version` → exit 127 and then answered "I cannot install
|
|
330
|
+
// system-level software like Rust on your host machine — I am physically
|
|
331
|
+
// unable to build the Tauri application", handing the user a manual
|
|
332
|
+
// `curl … | sh` step. It was not sandboxed at all; it had a real shell and
|
|
333
|
+
// the user had just granted terminal permission (trace
|
|
334
|
+
// trace-1791127992452-qzgodi). A missing prerequisite is a step to
|
|
335
|
+
// complete, not a capability limit — so this states that plainly.
|
|
336
|
+
'## Toolchains and missing prerequisites',
|
|
337
|
+
'- You run on the user\u2019s REAL machine with a REAL shell (run_terminal). You are NOT sandboxed and NOT "enclosed".',
|
|
338
|
+
'- If a command fails because a tool is missing (`command not found`, exit 127), INSTALL it yourself with run_terminal — e.g. `brew install <tool>`, `xcode-select --install`, the language installer (rustup for cargo/rust, nvm/npm for node, pyenv/pip for python, go, …), or apt/winget — then re-run the original command.',
|
|
339
|
+
'- NEVER answer that you are unable or not permitted to install a prerequisite, and never hand the user an install step you were asked to do. A missing tool is a step to complete, not a wall.',
|
|
340
|
+
'',
|
|
319
341
|
TOOL_CONTRACT_JSON,
|
|
320
342
|
].join('\n');
|
|
321
343
|
}
|
|
344
|
+
/**
|
|
345
|
+
* #30 — does this turn carry an HONESTY FLAG?
|
|
346
|
+
*
|
|
347
|
+
* A flagged answer must NEVER be written to the shared response cache. The cache
|
|
348
|
+
* stores text only, so a later identical prompt — on the CLI, the dashboard or the
|
|
349
|
+
* gateway — was served the reply with every flag absent: a known-unverified claim
|
|
350
|
+
* (`unverifiedActionClaim` / `unverifiedEditClaim` / `unverifiedBuildClaim`), an
|
|
351
|
+
* announced-but-unperformed action (`unfulfilledPromise`), a missing deliverable
|
|
352
|
+
* (`undeliveredArtifact`) or an inert turn (`noActionTaken`) replayed as a clean
|
|
353
|
+
* answer, on every surface at once. The flags exist because those answers must
|
|
354
|
+
* not be replayed as settled, so the turn is not cached and re-derives instead.
|
|
355
|
+
*/
|
|
356
|
+
export function turnCarriesHonestyFlag(result) {
|
|
357
|
+
return Boolean(result.unverifiedActionClaim ||
|
|
358
|
+
result.unfulfilledPromise ||
|
|
359
|
+
result.undeliveredArtifact ||
|
|
360
|
+
result.unverifiedBuildClaim ||
|
|
361
|
+
result.unverifiedEdit ||
|
|
362
|
+
result.unverifiedEditClaim ||
|
|
363
|
+
result.noActionTaken);
|
|
364
|
+
}
|
|
322
365
|
// ─── ChatCommand ────────────────────────────────────────────────────────────
|
|
323
366
|
export class ChatCommand extends BaseCommand {
|
|
324
367
|
devModeAuto = false;
|
|
@@ -364,7 +407,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
364
407
|
* console injects a per-session store instead; this is the CLI/execute
|
|
365
408
|
* default so a plan survives across turns within one chat session).
|
|
366
409
|
*/
|
|
367
|
-
planStore =
|
|
410
|
+
planStore = createPersistentPlanStore(`cli:${process.cwd()}`);
|
|
368
411
|
/**
|
|
369
412
|
* Whether the cold-start probe has fired this session. On a fresh registry
|
|
370
413
|
* (no verified models yet) the FIRST auto pick fires a background
|
|
@@ -435,16 +478,35 @@ export class ChatCommand extends BaseCommand {
|
|
|
435
478
|
// Best-effort — an unreadable config leaves the previous behavior.
|
|
436
479
|
}
|
|
437
480
|
}
|
|
438
|
-
|
|
481
|
+
// A8 — a CONCRETE provider with `-m auto` means that provider's own auto
|
|
482
|
+
// (e.g. an OmniRoute combo), NOT our auto-route. isAgentAutoRoute excludes
|
|
483
|
+
// that case so the pin reaches the provider instead of being re-routed.
|
|
484
|
+
let autoMode = isAgentAutoRoute(mergedOpts.provider, mergedOpts.model);
|
|
439
485
|
let { type, provider } = autoMode
|
|
440
486
|
? await this.getProvider({})
|
|
441
487
|
: await this.getProvider(mergedOpts);
|
|
442
488
|
let model = mergedOpts.model;
|
|
489
|
+
// B — captured from the route so the post-route capability gate below can
|
|
490
|
+
// judge the FINAL pair without re-resolving anything.
|
|
491
|
+
let routeVerdict;
|
|
492
|
+
let routedText;
|
|
443
493
|
if (autoMode) {
|
|
444
|
-
|
|
494
|
+
// Intent-aware escalation: a bare "yes"/"do it" continuing software work
|
|
495
|
+
// routes on the prior ask, not on the signal-free continuation. When
|
|
496
|
+
// there is no such hint the call is byte-identical to before.
|
|
497
|
+
const routingText = continuationSoftwareText(message, opts.history ?? []) ?? undefined;
|
|
498
|
+
const routed = routingText
|
|
499
|
+
? await this.routeMessageAuto(message, [], { routingText })
|
|
500
|
+
: await this.routeMessageAuto(message);
|
|
445
501
|
type = routed.type;
|
|
446
502
|
provider = routed.provider;
|
|
447
503
|
model = routed.model;
|
|
504
|
+
routedText = routingText;
|
|
505
|
+
routeVerdict = {
|
|
506
|
+
complexity: routed.complexity,
|
|
507
|
+
agenticCapable: routed.agenticCapable,
|
|
508
|
+
taskProfile: routed.taskProfile,
|
|
509
|
+
};
|
|
448
510
|
}
|
|
449
511
|
// P3 — tell the GUI where the turn is headed before the tool loop runs.
|
|
450
512
|
const isLocalFallback = autoMode && type === 'local';
|
|
@@ -452,6 +514,85 @@ export class ChatCommand extends BaseCommand {
|
|
|
452
514
|
? ' ⚠️ local model only — run `nuvira models` or `nuvira provider set` to add a cloud provider'
|
|
453
515
|
: '';
|
|
454
516
|
opts.onProgress?.(` 🧠 routed to ${provider.name}${model ? ` / ${model}` : ''} — working…${localWarning}`);
|
|
517
|
+
// ── Workstream B — agentic capability gate (consent-first) ──────────────
|
|
518
|
+
// The router RECORDED a verdict (A1); the turn must ACT on it. A software/
|
|
519
|
+
// agentic ask must never silently run on a weak model. Consent is per
|
|
520
|
+
// SESSION: one answer covers this session, and a new chat/task asks again.
|
|
521
|
+
//
|
|
522
|
+
// Interactive = an injected askUser (the dashboard console) or a real TTY.
|
|
523
|
+
// A piped/headless run never reaches the ask — it falls to the configured
|
|
524
|
+
// policy, whose default (`deny`/retry) can never silently downgrade.
|
|
525
|
+
if (autoMode && routeVerdict && routeVerdict.agenticCapable === false) {
|
|
526
|
+
const sessionId = opts.debugSession;
|
|
527
|
+
const interactive = Boolean(opts.askUser) || Boolean(process.stdin.isTTY);
|
|
528
|
+
const policy = resolveWeakModelPolicy(this.configManager);
|
|
529
|
+
const asDecision = {
|
|
530
|
+
complexity: routeVerdict.complexity,
|
|
531
|
+
taskProfile: (routeVerdict.taskProfile ?? {
|
|
532
|
+
intent: 'unknown',
|
|
533
|
+
requiresVerification: false,
|
|
534
|
+
}),
|
|
535
|
+
provider: type,
|
|
536
|
+
model: model ?? '',
|
|
537
|
+
agenticCapable: false,
|
|
538
|
+
};
|
|
539
|
+
let gate = assertAgenticRoute(asDecision, { sessionId, policy, interactive });
|
|
540
|
+
if (gate.action === 'ask') {
|
|
541
|
+
try {
|
|
542
|
+
const ask = opts.askUser ??
|
|
543
|
+
(await import('../tools/ask-user.js')).renderAskUser;
|
|
544
|
+
const answer = await ask(`This is a software task, but only a weak model is available ` +
|
|
545
|
+
`(${provider.name}${model ? ` / ${model}` : ''}). How should I proceed for this session?`, [
|
|
546
|
+
{ label: 'Approve the weak model for this session' },
|
|
547
|
+
{ label: 'Wait for a strong model only' },
|
|
548
|
+
], false);
|
|
549
|
+
const granted = Number(answer.index) === 0;
|
|
550
|
+
if (sessionId)
|
|
551
|
+
setWeakModelConsent(sessionId, granted ? 'granted' : 'denied');
|
|
552
|
+
gate = { ...gate, action: granted ? 'proceed-weak-consented' : 'retry-strong' };
|
|
553
|
+
}
|
|
554
|
+
catch {
|
|
555
|
+
// No answer reachable — treat as deny (never a silent downgrade).
|
|
556
|
+
gate = { ...gate, action: 'retry-strong' };
|
|
557
|
+
}
|
|
558
|
+
}
|
|
559
|
+
if (gate.action === 'retry-strong') {
|
|
560
|
+
// Try ONE re-route that EXCLUDES the weak provider; accept it only if
|
|
561
|
+
// the returned pair is genuinely agentic-capable. Nothing capable →
|
|
562
|
+
// refuse honestly rather than run the weak model against consent.
|
|
563
|
+
let next = null;
|
|
564
|
+
try {
|
|
565
|
+
next = routedText
|
|
566
|
+
? await this.routeMessageAuto(message, [type], { routingText: routedText })
|
|
567
|
+
: await this.routeMessageAuto(message, [type]);
|
|
568
|
+
}
|
|
569
|
+
catch {
|
|
570
|
+
next = null;
|
|
571
|
+
}
|
|
572
|
+
if (next && isAgenticCapableModel(next.model, next.type)) {
|
|
573
|
+
type = next.type;
|
|
574
|
+
provider = next.provider;
|
|
575
|
+
model = next.model;
|
|
576
|
+
opts.onProgress?.(` 🧠 re-routed to an agentic-capable model: ${provider.name}${model ? ` / ${model}` : ''}`);
|
|
577
|
+
}
|
|
578
|
+
else {
|
|
579
|
+
return {
|
|
580
|
+
content: gate.notice ??
|
|
581
|
+
'No agentic-capable model is available for this software task right now.',
|
|
582
|
+
followups: [],
|
|
583
|
+
generationFailed: true,
|
|
584
|
+
refused: true,
|
|
585
|
+
provider: type,
|
|
586
|
+
model,
|
|
587
|
+
transport: 'none',
|
|
588
|
+
};
|
|
589
|
+
}
|
|
590
|
+
}
|
|
591
|
+
else if (gate.notice) {
|
|
592
|
+
// proceed-weak-consented — say so plainly; never a silent weak model.
|
|
593
|
+
opts.onProgress?.(` ${gate.notice}`);
|
|
594
|
+
}
|
|
595
|
+
}
|
|
455
596
|
// P4 — when a project is attached, recall its prior sessions + facts
|
|
456
597
|
// FRESH per turn (the snapshot is cached, the recall is not — prior work
|
|
457
598
|
// may have landed since the last turn). Best-effort: empty recall injects
|
|
@@ -513,11 +654,48 @@ export class ChatCommand extends BaseCommand {
|
|
|
513
654
|
...this.turnEnvelopeOf(answer),
|
|
514
655
|
};
|
|
515
656
|
}
|
|
657
|
+
// E1 — assemble the turn report from RECORDED evidence (plan store, tool
|
|
658
|
+
// outcomes, honesty flags). Derived, never narrated, so the trust verdict
|
|
659
|
+
// cannot be talked up by the model.
|
|
660
|
+
let turnReport;
|
|
661
|
+
try {
|
|
662
|
+
const planSnapshot = (opts.planStore ?? this.planStore).snapshot?.() ?? null;
|
|
663
|
+
turnReport = buildTurnReport({
|
|
664
|
+
goal: message,
|
|
665
|
+
plan: planSnapshot,
|
|
666
|
+
toolCalls: answer.toolCalls,
|
|
667
|
+
successfulToolCalls: answer.successfulToolCalls,
|
|
668
|
+
mutations: answer.runTrace?.mutations,
|
|
669
|
+
changedPaths: answer.runTrace?.paths,
|
|
670
|
+
flags: {
|
|
671
|
+
unverifiedActionClaim: answer.unverifiedActionClaim,
|
|
672
|
+
unverifiedEdit: answer.unverifiedEdit,
|
|
673
|
+
unverifiedEditClaim: answer.unverifiedEditClaim,
|
|
674
|
+
unverifiedBuildClaim: answer.unverifiedBuildClaim,
|
|
675
|
+
undeliveredArtifact: answer.undeliveredArtifact,
|
|
676
|
+
unfulfilledPromise: answer.unfulfilledPromise,
|
|
677
|
+
noActionTaken: answer.noActionTaken,
|
|
678
|
+
},
|
|
679
|
+
});
|
|
680
|
+
}
|
|
681
|
+
catch {
|
|
682
|
+
// A report must never break the turn.
|
|
683
|
+
}
|
|
684
|
+
// E-trace — persist the report on the turn's reasoning trace so it is
|
|
685
|
+
// reviewable after the fact (the Trace tab renders it), not only in this
|
|
686
|
+
// turn's return value. Best-effort: a trace write never breaks a turn.
|
|
687
|
+
recordTurnReport(answer.traceId, turnReport);
|
|
688
|
+
// E3 — surface a non-trivial report on the console. A plain answer (no
|
|
689
|
+
// plan, nothing changed) produces no summary and stays silent.
|
|
690
|
+
if (turnReport?.summary) {
|
|
691
|
+
opts.onProgress?.(formatTurnReport(turnReport));
|
|
692
|
+
}
|
|
516
693
|
// E3b: strip raw suggest_followups JSON embedded in content by the model
|
|
517
694
|
const cleanContent = stripToolCallArtifacts(answer.content || '');
|
|
518
695
|
return {
|
|
519
696
|
content: cleanContent,
|
|
520
697
|
followups: answer.followups ?? [],
|
|
698
|
+
...(turnReport ? { turnReport } : {}),
|
|
521
699
|
generationFailed: answer.generationFailed,
|
|
522
700
|
cancelled: answer.cancelled,
|
|
523
701
|
bounded: answer.bounded,
|
|
@@ -564,8 +742,8 @@ export class ChatCommand extends BaseCommand {
|
|
|
564
742
|
.description('Start an interactive chat session with AI')
|
|
565
743
|
.argument('[prompt]', 'Optional initial prompt')
|
|
566
744
|
.option('-f, --file <path>', 'Include file content as context')
|
|
567
|
-
.option('-p, --provider <provider>', 'Inference provider')
|
|
568
|
-
.option('-m, --model <model>', 'Model to
|
|
745
|
+
.option('-p, --provider <provider>', 'Inference provider to pin this chat to')
|
|
746
|
+
.option('-m, --model <model>', 'Model to pin this chat to (if omitted, an interactive picker will appear)')
|
|
569
747
|
.option('--no-cache', 'Disable response caching')
|
|
570
748
|
.option('-d, --dev', 'Always dispatch requests to the coding pipeline (no confirmation)', false)
|
|
571
749
|
// WS5 (#27) — isolation and partial resume, as the two things an operator
|
|
@@ -581,6 +759,12 @@ export class ChatCommand extends BaseCommand {
|
|
|
581
759
|
.option('--worktree', 'Run this turn in its own git worktree of the project and report the diff against the base commit. Refuses rather than running unisolated when the directory cannot be isolated (also asked for by NUVIRA_ISOLATE=1)')
|
|
582
760
|
.option('--keep-worktree', 'Keep the isolated worktree after the turn instead of removing it')
|
|
583
761
|
.option('--resume [id]', 'Replay the recorded steps of this ask whose input is unchanged instead of paying for them again (defaults to the record for this goal + directory; also asked for by NUVIRA_RESUME=1)')
|
|
762
|
+
.addHelpText('after', '\nModel routing:\n' +
|
|
763
|
+
' Pinning -p/--provider (and -m/--model) does NOT stop the router from falling\n' +
|
|
764
|
+
' over to another model if the pinned one is unavailable — auto routing takes\n' +
|
|
765
|
+
' over and the turn says so. To work with the pinned model ONLY, set\n' +
|
|
766
|
+
' NUVIRA_STRICT_MODEL=1: a dead pin then fails with a message naming the pair\n' +
|
|
767
|
+
' instead of substituting another model.\n')
|
|
584
768
|
.action(async (prompt, options) => {
|
|
585
769
|
await this.execute(prompt, options || {});
|
|
586
770
|
});
|
|
@@ -608,7 +792,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
608
792
|
const activeOpts = applyActiveModel({ provider: options?.provider, model: options?.model });
|
|
609
793
|
const mergedOpts = { ...options, provider: activeOpts.provider, model: activeOpts.model };
|
|
610
794
|
// ── Auto routing mode: agent decides the best provider/model per message ──
|
|
611
|
-
|
|
795
|
+
// A8 — a pinned concrete provider with `-m auto` is the provider's own auto,
|
|
796
|
+
// not ours. See isAgentAutoRoute.
|
|
797
|
+
let autoMode = isAgentAutoRoute(mergedOpts.provider, mergedOpts.model);
|
|
612
798
|
let { type, provider } = autoMode
|
|
613
799
|
? await this.getProvider({})
|
|
614
800
|
: await this.getProvider(mergedOpts);
|
|
@@ -812,7 +998,12 @@ export class ChatCommand extends BaseCommand {
|
|
|
812
998
|
// so context-fit routing reacts to a long session, not just the task
|
|
813
999
|
// text. Estimation only — never a hard block.
|
|
814
1000
|
const historyEstimate = estimateTokens(history.map((h) => h.content).join('\n') + '\n' + message);
|
|
815
|
-
|
|
1001
|
+
// Intent-aware escalation for a bare continuation of software work.
|
|
1002
|
+
const routingText = continuationSoftwareText(message, history) ?? undefined;
|
|
1003
|
+
const routed = await this.routeMessageAuto(message, [], {
|
|
1004
|
+
contextHintTokens: historyEstimate,
|
|
1005
|
+
...(routingText ? { routingText } : {}),
|
|
1006
|
+
});
|
|
816
1007
|
type = routed.type;
|
|
817
1008
|
provider = routed.provider;
|
|
818
1009
|
effectiveModel = routed.model;
|
|
@@ -974,25 +1165,35 @@ export class ChatCommand extends BaseCommand {
|
|
|
974
1165
|
const cacheModel = this.cacheModelFor(session);
|
|
975
1166
|
if (cacheEnabled) {
|
|
976
1167
|
try {
|
|
977
|
-
|
|
978
|
-
|
|
1168
|
+
// #30 — read the entry WITH its recorded activity, not just the text: a
|
|
1169
|
+
// replay that dropped `toolCalls` rendered no tool cards on the dashboard
|
|
1170
|
+
// while the first run did, so a repeated prompt read as a turn that did
|
|
1171
|
+
// nothing. The text alone is still what the answer is; the activity is
|
|
1172
|
+
// reported so the surface is honest about what the cached turn DID.
|
|
1173
|
+
const cached = await cache.getEntry(message, cacheModel, session.type, turnScope);
|
|
1174
|
+
if (cached) {
|
|
979
1175
|
// NOTE: the cached answer is NOT printed here — the caller prints
|
|
980
1176
|
// content AFTER runChatAnswer returns (answer-first ordering). A
|
|
981
1177
|
// print here would show the answer before the turn's own progress
|
|
982
1178
|
// lines AND double-print it.
|
|
983
1179
|
history.push({ role: 'user', content: message });
|
|
984
|
-
history.push({ role: 'assistant', content:
|
|
985
|
-
this.memoryNoteTurn(message,
|
|
1180
|
+
history.push({ role: 'assistant', content: cached.response });
|
|
1181
|
+
this.memoryNoteTurn(message, cached.response);
|
|
986
1182
|
// WS2 — a cache replay reached no model, so the log says exactly that
|
|
987
1183
|
// rather than borrowing an attribution from a turn that did not run.
|
|
988
1184
|
// The workspace the replayed answer belongs to is recorded with the hit:
|
|
989
1185
|
// a cache replay does no work, so "which project is this answer about?"
|
|
990
1186
|
// is the one fact needed to tell a replay from a real turn.
|
|
991
|
-
debugLog?.event('cache.hit', { chars:
|
|
1187
|
+
debugLog?.event('cache.hit', { chars: cached.response.length, scope: turnScope });
|
|
992
1188
|
const cacheNotice = debugLogNotice(ctxOverrides?.debugSurface ?? 'cli-chat', debugLog?.write() ?? null);
|
|
993
1189
|
if (cacheNotice)
|
|
994
1190
|
logger.info(cacheNotice);
|
|
995
|
-
return {
|
|
1191
|
+
return {
|
|
1192
|
+
content: cached.response,
|
|
1193
|
+
...(cached.toolCalls ? { toolCalls: cached.toolCalls } : {}),
|
|
1194
|
+
...(cached.successfulToolCalls ? { successfulToolCalls: cached.successfulToolCalls } : {}),
|
|
1195
|
+
...(cached.bounded ? { bounded: cached.bounded } : {}),
|
|
1196
|
+
};
|
|
996
1197
|
}
|
|
997
1198
|
}
|
|
998
1199
|
catch {
|
|
@@ -1112,17 +1313,23 @@ export class ChatCommand extends BaseCommand {
|
|
|
1112
1313
|
// computed for the generation-FAILED fallback only — it does NOT reach the
|
|
1113
1314
|
// prompt, so on this surface the MODEL decides and the rules are invisible.
|
|
1114
1315
|
const systemText = buildToolSystemPrompt(parsed);
|
|
1115
|
-
//
|
|
1116
|
-
//
|
|
1117
|
-
//
|
|
1118
|
-
//
|
|
1119
|
-
//
|
|
1120
|
-
//
|
|
1121
|
-
//
|
|
1316
|
+
// Skill hint — MODE-DEPENDENT (see resolveSkillHintMode):
|
|
1317
|
+
// - `match` (default) — the small keyword-matched hint: ONE skill, and
|
|
1318
|
+
// only when the goal really matches; otherwise nothing. This is the
|
|
1319
|
+
// 3.3.10 behaviour and keeps the prompt small.
|
|
1320
|
+
// - `catalog` (opt-in) — the full name+description catalog, which lets the
|
|
1321
|
+
// MODEL pick a skill but costs ~24K chars on every turn, so it must be
|
|
1322
|
+
// chosen (`NUVIRA_SKILL_CATALOG=catalog` or `skills.catalogHint`).
|
|
1323
|
+
// - `off` — never inject one.
|
|
1324
|
+
// Best-effort: any failure returns '' and the turn proceeds byte-identically.
|
|
1122
1325
|
let skillHint = '';
|
|
1123
1326
|
try {
|
|
1124
|
-
const {
|
|
1125
|
-
|
|
1327
|
+
const { buildConfiguredSkillHint, markLoopSkillUsed } = await import('../tools/loop-skill-hint.js');
|
|
1328
|
+
const injected = {
|
|
1329
|
+
value: null,
|
|
1330
|
+
};
|
|
1331
|
+
skillHint = await buildConfiguredSkillHint(message, this.configManager, injected);
|
|
1332
|
+
await markLoopSkillUsed(injected.value);
|
|
1126
1333
|
}
|
|
1127
1334
|
catch {
|
|
1128
1335
|
skillHint = ''; // best-effort — a hint failure never breaks the turn
|
|
@@ -1281,8 +1488,23 @@ export class ChatCommand extends BaseCommand {
|
|
|
1281
1488
|
}
|
|
1282
1489
|
}
|
|
1283
1490
|
// P0.7 — forward plan mutations to the GUI (structured checklist).
|
|
1284
|
-
if (
|
|
1285
|
-
|
|
1491
|
+
if (event === 'plan:changed') {
|
|
1492
|
+
const snapshot = data;
|
|
1493
|
+
if (ctxOverrides?.onPlanChange) {
|
|
1494
|
+
ctxOverrides.onPlanChange(snapshot);
|
|
1495
|
+
}
|
|
1496
|
+
else {
|
|
1497
|
+
// No GUI consumer (the interactive CLI): show the PROGRESS TABLE in
|
|
1498
|
+
// the terminal so a user watching the run sees the plan advance,
|
|
1499
|
+
// not just the model's narration. A settled plan prints its
|
|
1500
|
+
// achieved SUMMARY instead — the same text the turn closes on.
|
|
1501
|
+
const store = ctxOverrides?.planStore ?? this.planStore;
|
|
1502
|
+
const settled = snapshot.steps.length > 0 &&
|
|
1503
|
+
snapshot.steps.every((s) => s.status === 'done' || s.status === 'blocked');
|
|
1504
|
+
const rendered = settled ? store.summary?.() : store.toTable?.();
|
|
1505
|
+
if (rendered)
|
|
1506
|
+
logger.info(`\n${rendered}\n`);
|
|
1507
|
+
}
|
|
1286
1508
|
}
|
|
1287
1509
|
// P3b — forward git diff payloads to the GUI (the diff card).
|
|
1288
1510
|
if (ctxOverrides?.onGitDiff && event === 'git:diff') {
|
|
@@ -1332,6 +1554,13 @@ export class ChatCommand extends BaseCommand {
|
|
|
1332
1554
|
// default interactive renderer.
|
|
1333
1555
|
...(ctxOverrides?.askUser ? { askUser: ctxOverrides.askUser } : {}),
|
|
1334
1556
|
...(ctxOverrides?.gateway ? { gateway: ctxOverrides.gateway } : {}),
|
|
1557
|
+
// P0.7 — the plan store. This was DROPPED here: `answerOnce` put a
|
|
1558
|
+
// per-session store on `ctxOverrides.planStore`, but only askUser/gateway
|
|
1559
|
+
// were threaded into the tool context, so every surface fell back to the
|
|
1560
|
+
// shared module store — which is why plans leaked across sessions and a
|
|
1561
|
+
// reload started a blank checklist. Thread it (per-session when injected,
|
|
1562
|
+
// else this command's project-scoped store).
|
|
1563
|
+
planStore: ctxOverrides?.planStore ?? this.planStore,
|
|
1335
1564
|
// C2 verify with the actual session model (verify_requirement tool).
|
|
1336
1565
|
callLLM: (prompt, opts) => session.provider.generate(prompt, { ...opts, model: session.model }),
|
|
1337
1566
|
// I3: tools that return {artifact, result} deliverables are recorded to
|
|
@@ -1349,7 +1578,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
1349
1578
|
},
|
|
1350
1579
|
},
|
|
1351
1580
|
};
|
|
1352
|
-
const callModel = this.buildToolCallModel(message, session, options, mode, ctxOverrides?.onToken, ctxOverrides?.signal);
|
|
1581
|
+
const callModel = this.buildToolCallModel(message, session, options, mode, ctxOverrides?.onToken, ctxOverrides?.signal, continuationSoftwareText(message, history) ?? undefined);
|
|
1353
1582
|
let result;
|
|
1354
1583
|
// v1.8x audit — CHAT TRACE CAPTURE: every LLM call in a chat turn is now
|
|
1355
1584
|
// recorded to ~/.nuvira/memory/reasoning-traces.json (source 'chat'), so
|
|
@@ -1367,6 +1596,64 @@ export class ChatCommand extends BaseCommand {
|
|
|
1367
1596
|
});
|
|
1368
1597
|
// G18 — the tool-context emit (declared above) now has somewhere to write.
|
|
1369
1598
|
traceIdForEvents = chatTraceId;
|
|
1599
|
+
// A2 — record the routing DECISION as a first-class event, so a turn that
|
|
1600
|
+
// ran on a weak/incapable model is self-evident in the Trace tab. The
|
|
1601
|
+
// failed Tauri turn had no such record; this is the instrument that would
|
|
1602
|
+
// have shown `agenticCapable:false` at the moment of the choice.
|
|
1603
|
+
if (this.lastRouteSnapshot) {
|
|
1604
|
+
const snap = this.lastRouteSnapshot;
|
|
1605
|
+
recordTraceEvent(chatTraceId, {
|
|
1606
|
+
kind: 'decision',
|
|
1607
|
+
gate: 'routing',
|
|
1608
|
+
summary: `routed to ${snap.provider}/${snap.model} ` +
|
|
1609
|
+
`(complexity ${snap.complexity}, score ${snap.score.toFixed(3)})` +
|
|
1610
|
+
(snap.agenticCapable === false
|
|
1611
|
+
? ` — NOT agentic-capable${snap.overrideReason ? ` (${snap.overrideReason})` : ''}`
|
|
1612
|
+
: ''),
|
|
1613
|
+
routing: snap,
|
|
1614
|
+
});
|
|
1615
|
+
}
|
|
1616
|
+
// D1 — measure the OUTBOUND context and, past the ceiling, degrade the
|
|
1617
|
+
// lowest-value optional contributor first (skill hint → work digest →
|
|
1618
|
+
// recall → …). The identity/tool contract is never trimmed. This is the
|
|
1619
|
+
// guard the 3.3.11 bloat (7.6K → 32.8K chars) never had.
|
|
1620
|
+
try {
|
|
1621
|
+
const contextBudget = resolvePromptBudget(this.configManager);
|
|
1622
|
+
const historyChars = history.reduce((n, h) => n + (h.content?.length ?? 0), 0);
|
|
1623
|
+
const report = measurePromptBudget([
|
|
1624
|
+
{ name: 'system:identity+tool-contract', chars: systemText.length },
|
|
1625
|
+
{ name: 'system:channel-policy', chars: systemPolicyBlock.length },
|
|
1626
|
+
{ name: 'skill-hint', chars: skillHint.length, dropPriority: 10 },
|
|
1627
|
+
{ name: 'working-state', chars: workingStateBlock.length, dropPriority: 25 },
|
|
1628
|
+
{ name: 'recall', chars: ctxOverrides?.recallContext?.length ?? 0, dropPriority: 30 },
|
|
1629
|
+
{
|
|
1630
|
+
name: 'project-context',
|
|
1631
|
+
chars: ctxOverrides?.projectContext?.length ?? ambientProjectContext?.length ?? 0,
|
|
1632
|
+
dropPriority: 40,
|
|
1633
|
+
},
|
|
1634
|
+
{ name: 'file-context', chars: fileContext?.length ?? 0, dropPriority: 45 },
|
|
1635
|
+
{ name: 'history+ask', chars: historyChars },
|
|
1636
|
+
], { budget: contextBudget });
|
|
1637
|
+
// Apply the FIRST ladder step (the documented, safe one): drop the skill
|
|
1638
|
+
// hint from the assembled system message.
|
|
1639
|
+
if (report.trims.includes('skill-hint') && skillHint) {
|
|
1640
|
+
thread[0] = { role: 'system', content: systemText + systemPolicyBlock };
|
|
1641
|
+
}
|
|
1642
|
+
if (report.level !== 'ok') {
|
|
1643
|
+
recordTraceEvent(chatTraceId, {
|
|
1644
|
+
kind: 'decision',
|
|
1645
|
+
gate: 'context-budget',
|
|
1646
|
+
summary: formatPromptBudgetBreakdown(report) +
|
|
1647
|
+
(report.trims.length ? ` — trimmed: ${report.trims.join(', ')}` : ''),
|
|
1648
|
+
});
|
|
1649
|
+
}
|
|
1650
|
+
}
|
|
1651
|
+
catch {
|
|
1652
|
+
// A budget measurement must never break the turn.
|
|
1653
|
+
}
|
|
1654
|
+
// The instant this turn's model walk began. A failed generation records the
|
|
1655
|
+
// WALK (below) relative to this mark, so the trace can say what was tried.
|
|
1656
|
+
const modelWalkMark = Date.now();
|
|
1370
1657
|
const seenStepDigests = new Set();
|
|
1371
1658
|
const digestPrompt = (p) => {
|
|
1372
1659
|
try {
|
|
@@ -1498,6 +1785,36 @@ export class ChatCommand extends BaseCommand {
|
|
|
1498
1785
|
// (`answerQualityError`), which the loop rethrows once every candidate has
|
|
1499
1786
|
// narrated.
|
|
1500
1787
|
logger.error(String(err));
|
|
1788
|
+
// DIAGNOSABILITY — a failed turn used to record only the LAST attempt's
|
|
1789
|
+
// provider/model beside the FIRST error, so a reader could not tell which
|
|
1790
|
+
// model actually ran, nor WHY other models were not used. Record the WALK:
|
|
1791
|
+
// what was tried, what was parked (and for how long), and how large the
|
|
1792
|
+
// eligible pool was — the facts that separate a real shortage from a
|
|
1793
|
+
// routing gap. This is the exact ambiguity in the traces that prompted it:
|
|
1794
|
+
// a step labelled `local/qwen2.5:0.5b` carrying Gemini's 429.
|
|
1795
|
+
try {
|
|
1796
|
+
const report = modelBreadthReport(modelWalkMark, this.configManager);
|
|
1797
|
+
const tried = report.tried.filter((a) => !a.skipped);
|
|
1798
|
+
const parked = report.parked.filter((r) => r.active);
|
|
1799
|
+
const triedList = tried
|
|
1800
|
+
.slice(0, 6)
|
|
1801
|
+
.map((a) => `${a.provider}/${a.model} (${a.reason})`)
|
|
1802
|
+
.join(', ');
|
|
1803
|
+
const parkedList = parked
|
|
1804
|
+
.slice(0, 6)
|
|
1805
|
+
.map((r) => `${r.provider}${r.model ? `/${r.model}` : ''} (${r.kind})`)
|
|
1806
|
+
.join(', ');
|
|
1807
|
+
recordTraceEvent(chatTraceId, {
|
|
1808
|
+
kind: 'failover',
|
|
1809
|
+
summary: `the model layer failed — eligible pool ${report.poolSize ?? '?'} model(s) across ` +
|
|
1810
|
+
`${report.poolProviders ?? '?'} provider(s), ${tried.length} tried, ${parked.length} parked` +
|
|
1811
|
+
(triedList ? `; tried: ${triedList}` : '') +
|
|
1812
|
+
(parkedList ? `; parked: ${parkedList}` : ''),
|
|
1813
|
+
});
|
|
1814
|
+
}
|
|
1815
|
+
catch {
|
|
1816
|
+
// Diagnosis is a courtesy — it must never break the failure path.
|
|
1817
|
+
}
|
|
1501
1818
|
endTrace(chatTraceId, false, { kind: 'failed' });
|
|
1502
1819
|
result = {
|
|
1503
1820
|
// Sanitized on purpose: this content is delivered verbatim by every
|
|
@@ -1609,11 +1926,29 @@ export class ChatCommand extends BaseCommand {
|
|
|
1609
1926
|
if (result.content.trim() && !result.generationFailed && !result.cancelled) {
|
|
1610
1927
|
if (cacheEnabled) {
|
|
1611
1928
|
try {
|
|
1612
|
-
//
|
|
1613
|
-
//
|
|
1614
|
-
//
|
|
1615
|
-
//
|
|
1616
|
-
|
|
1929
|
+
// #30 — NEVER cache a turn that carries an honesty flag. The cache
|
|
1930
|
+
// stores text only, so storing a flagged answer would let a later
|
|
1931
|
+
// identical prompt (or the same prompt on another surface) replay a
|
|
1932
|
+
// known-unverified claim as a clean one — a truthfulness hole the whole
|
|
1933
|
+
// flag system exists to close. A flagged turn re-derives instead.
|
|
1934
|
+
if (turnCarriesHonestyFlag(result)) {
|
|
1935
|
+
debugLog?.event('cache.skip', { reason: 'honesty-flag', scope: turnScope });
|
|
1936
|
+
}
|
|
1937
|
+
else {
|
|
1938
|
+
// Keyed by the model that ACTUALLY answered (tryGenerate records it
|
|
1939
|
+
// on success), so a weak model's reply is never replayed as a strong
|
|
1940
|
+
// model's. `cacheModel` is the pre-flight fallback for the paths that
|
|
1941
|
+
// never resolve one (e.g. a cached-hit turn).
|
|
1942
|
+
//
|
|
1943
|
+
// #30 — the turn ACTIVITY rides with the text so a replay reports what
|
|
1944
|
+
// the cached turn did (tool cards, bounded), rather than reading as a
|
|
1945
|
+
// turn that did nothing.
|
|
1946
|
+
await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type, undefined, turnScope, {
|
|
1947
|
+
...(result.toolCalls ? { toolCalls: result.toolCalls } : {}),
|
|
1948
|
+
...(result.successfulToolCalls ? { successfulToolCalls: result.successfulToolCalls } : {}),
|
|
1949
|
+
...(result.bounded ? { bounded: true } : {}),
|
|
1950
|
+
});
|
|
1951
|
+
}
|
|
1617
1952
|
}
|
|
1618
1953
|
catch {
|
|
1619
1954
|
// Best-effort.
|
|
@@ -1710,6 +2045,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1710
2045
|
transport: result.transport,
|
|
1711
2046
|
// WS1 — the findings this turn recorded, with their verdicts.
|
|
1712
2047
|
findings,
|
|
2048
|
+
// E — the honest "what happened" facts the TurnReport is derived from.
|
|
2049
|
+
successfulToolCalls: result.successfulToolCalls,
|
|
2050
|
+
runTrace: result.runTrace,
|
|
2051
|
+
unverifiedEdit: result.unverifiedEdit,
|
|
2052
|
+
unverifiedEditClaim: result.unverifiedEditClaim,
|
|
2053
|
+
noActionTaken: result.noActionTaken,
|
|
2054
|
+
// E-trace — the trace this turn was recorded under, so `answerOnce` can
|
|
2055
|
+
// attach the TurnReport to it (see `recordTurnReport`).
|
|
2056
|
+
traceId: chatTraceId,
|
|
1713
2057
|
});
|
|
1714
2058
|
}
|
|
1715
2059
|
/**
|
|
@@ -1743,7 +2087,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1743
2087
|
return `${session.type}:unresolved`;
|
|
1744
2088
|
}
|
|
1745
2089
|
}
|
|
1746
|
-
buildToolCallModel(message, session, options, mode, onToken, signal
|
|
2090
|
+
buildToolCallModel(message, session, options, mode, onToken, signal,
|
|
2091
|
+
/**
|
|
2092
|
+
* Intent-aware failover: the prior software ask when this turn is a bare
|
|
2093
|
+
* continuation ("yes", "do it"), computed by the caller from the history
|
|
2094
|
+
* it owns. The initial route used it; the mid-turn failover walk needs it
|
|
2095
|
+
* too, or a continuation whose first cloud provider dies re-routes on the
|
|
2096
|
+
* signal-free message and can land on a tiny local model.
|
|
2097
|
+
*/
|
|
2098
|
+
routingText) {
|
|
1747
2099
|
return async (messages, schemas, stepOnToken, stepSignal) => {
|
|
1748
2100
|
// The effective token sink: the caller's stream wins; when a step-level
|
|
1749
2101
|
// sink is also given (loop passthrough) they are the same channel.
|
|
@@ -1802,6 +2154,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1802
2154
|
// refused it). The reason names the provider and the rule.
|
|
1803
2155
|
throw new Error(`Governance policy: ${verdict.reason}`);
|
|
1804
2156
|
}
|
|
2157
|
+
// ADMIN BUDGET (pinned path). The pin still runs the model the user
|
|
2158
|
+
// chose — this only REFUSES it when the user's own declared budget
|
|
2159
|
+
// (routing.quota window, admin cost cap) is already exceeded, instead
|
|
2160
|
+
// of spending past the control they set in the dashboard. A no-op when
|
|
2161
|
+
// no budget is configured.
|
|
2162
|
+
const budget = adminBudgetVerdict(this.configManager, session.type, { model: session.model });
|
|
2163
|
+
if (!budget.allowed) {
|
|
2164
|
+
throw new Error(`Admin budget: ${budget.reason}`);
|
|
2165
|
+
}
|
|
1805
2166
|
}
|
|
1806
2167
|
const resolveEffectiveModel = (providerType, requested) => {
|
|
1807
2168
|
if (requested && requested !== 'default')
|
|
@@ -1953,13 +2314,27 @@ export class ChatCommand extends BaseCommand {
|
|
|
1953
2314
|
// success log per landed candidate.
|
|
1954
2315
|
logger.warn(` ⚠️ ${session.provider.name} failed — trying the next auto candidate...`);
|
|
1955
2316
|
const failed = new Set([session.type]);
|
|
2317
|
+
// Intent-aware escalation for the FAILOVER walk (the live junk path):
|
|
2318
|
+
// a mid-turn failure used to re-route on the bare message alone, so a
|
|
2319
|
+
// software turn whose first cloud provider died fell through to a tiny
|
|
2320
|
+
// local model (`local/qwen2.5:0.5b`) that FABRICATED tool output
|
|
2321
|
+
// (trace-1791118650644-d73hyr). Pass the same prior-software-ask hint
|
|
2322
|
+
// the initial route used, so the router's agentic capability floor
|
|
2323
|
+
// applies here too — but NEVER drop a candidate the floor would keep,
|
|
2324
|
+
// so auto still cannot dead-end (local stays the last resort).
|
|
2325
|
+
const failoverRoutingText = routingText;
|
|
1956
2326
|
// Try ALL ranked candidates (no 3-candidate cap) — bounded by
|
|
1957
2327
|
// the number of known providers to prevent infinite loops.
|
|
1958
2328
|
const maxAttempts = 10;
|
|
1959
2329
|
for (let i = 0; i < maxAttempts; i++) {
|
|
1960
2330
|
let next = null;
|
|
1961
2331
|
try {
|
|
1962
|
-
next =
|
|
2332
|
+
next = failoverRoutingText
|
|
2333
|
+
? await this.routeMessageAuto(message, [...failed], {
|
|
2334
|
+
routingText: failoverRoutingText,
|
|
2335
|
+
fallbackFrom: firstType,
|
|
2336
|
+
})
|
|
2337
|
+
: await this.routeMessageAuto(message, [...failed], { fallbackFrom: firstType });
|
|
1963
2338
|
}
|
|
1964
2339
|
catch {
|
|
1965
2340
|
break;
|
|
@@ -1967,6 +2342,17 @@ export class ChatCommand extends BaseCommand {
|
|
|
1967
2342
|
if (!next || next.type === session.type || failed.has(next.type))
|
|
1968
2343
|
break;
|
|
1969
2344
|
failed.add(next.type);
|
|
2345
|
+
// Last-resort truthfulness: if even the agentic floor could not
|
|
2346
|
+
// avoid a weak model, say so ONCE so a degraded answer is never
|
|
2347
|
+
// mistaken for a real one (the tiny local model fabricated tool
|
|
2348
|
+
// results instead of admitting it could not run them). C2 — the
|
|
2349
|
+
// wording comes from the SHARED helper so chat, the orchestrator
|
|
2350
|
+
// and the dashboard cannot describe this three different ways.
|
|
2351
|
+
if (!isAgenticCapableModel(next.model, next.type)) {
|
|
2352
|
+
const notice = weakRouteNotice({ provider: next.type, model: next.model }, true) ??
|
|
2353
|
+
`⚠️ no agentic-capable model left — falling back to ${next.type}/${next.model}`;
|
|
2354
|
+
logger.warn(` no agentic-capable model left — ${notice}`);
|
|
2355
|
+
}
|
|
1970
2356
|
// Opt-in confirmation (routing.promptOnFailover): 'manual' stops
|
|
1971
2357
|
// the walk and lets the caller's error recovery handle it. Gated on
|
|
1972
2358
|
// an interactive stdin (inherited from the shared single-shot
|
|
@@ -1999,8 +2385,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1999
2385
|
}
|
|
2000
2386
|
}
|
|
2001
2387
|
}
|
|
2002
|
-
else if (isRetryableError(classifyFallbackError(err))) {
|
|
2388
|
+
else if (!strictModelMode() && isRetryableError(classifyFallbackError(err))) {
|
|
2003
2389
|
// Non-auto: walk the shared fallback chain (retryable errors only).
|
|
2390
|
+
//
|
|
2391
|
+
// Under strict model mode the walk is exactly the substitution the
|
|
2392
|
+
// user forbade: `strictModelMode()` short-circuits this branch, so a
|
|
2393
|
+
// pinned model that cannot answer surfaces its own error instead of
|
|
2394
|
+
// quietly continuing on a provider the user did not choose (A2 — the
|
|
2395
|
+
// same defect the loop engine had, fixed for dashboard + CLI chat
|
|
2396
|
+
// here). The `tryGenerate` call above already reports the raw error.
|
|
2004
2397
|
// Providers the admin policy rules out are collected here so the
|
|
2005
2398
|
// failure can name POLICY as the reason instead of implying the model
|
|
2006
2399
|
// was unreachable.
|
|
@@ -2218,6 +2611,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
2218
2611
|
* so Auto routing never sends a request to a provider that would 401.
|
|
2219
2612
|
*/
|
|
2220
2613
|
async routeMessageAuto(message, excludeProviders = [], opts) {
|
|
2614
|
+
// The text ROUTING is decided from — the prior software ask for a bare
|
|
2615
|
+
// continuation, otherwise the message itself. `message` is used everywhere
|
|
2616
|
+
// else (the answer, the tool loop).
|
|
2617
|
+
const taskText = opts?.routingText?.trim() ? opts.routingText : message;
|
|
2221
2618
|
// Feed the SHARED circuit breaker into the router so a provider that has
|
|
2222
2619
|
// failed repeatedly (recorded by recordFailure below) is deprioritized by
|
|
2223
2620
|
// scoring, not just skipped by the candidate walk.
|
|
@@ -2235,7 +2632,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2235
2632
|
// circuit-breaker state on top.
|
|
2236
2633
|
// C3: the NLU parser seeds the router task-intent (same vocabulary every
|
|
2237
2634
|
// action command derives from resolveDispatch) when confident.
|
|
2238
|
-
const parsed = parseRequestSync(
|
|
2635
|
+
const parsed = parseRequestSync(taskText);
|
|
2239
2636
|
const dispatch = resolveDispatch(parsed);
|
|
2240
2637
|
// Routing decision cache (assessment v4 Phase 2): the loop engine
|
|
2241
2638
|
// resolves per turn; turns with identical STABLE routing inputs (intent,
|
|
@@ -2259,7 +2656,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2259
2656
|
const cacheSignature = routingCacheSignature([
|
|
2260
2657
|
'chat',
|
|
2261
2658
|
dispatch.taskIntentHint ?? null,
|
|
2262
|
-
analyzeComplexity(
|
|
2659
|
+
analyzeComplexity(taskText),
|
|
2263
2660
|
routingCfg.preferenceMode ?? null,
|
|
2264
2661
|
routingCfg.bandit === false ? 'b' : 'B',
|
|
2265
2662
|
routingCfg.mlRouter === true ? 'm' : 'M',
|
|
@@ -2272,7 +2669,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2272
2669
|
.sort()
|
|
2273
2670
|
.join(','),
|
|
2274
2671
|
]);
|
|
2275
|
-
const decision = withRoutingCache(cacheSignature, 30_000, () => getAutoRouter().resolve('chat',
|
|
2672
|
+
const decision = withRoutingCache(cacheSignature, 30_000, () => getAutoRouter().resolve('chat', taskText, {
|
|
2276
2673
|
...buildAutoResolveOptions(this.configManager, {
|
|
2277
2674
|
verbose: envBuff('DEBUG') === 'true',
|
|
2278
2675
|
contextHintTokens: opts?.contextHintTokens,
|
|
@@ -2288,6 +2685,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
2288
2685
|
score: decision.score,
|
|
2289
2686
|
complexity: String(decision.complexity),
|
|
2290
2687
|
explanation: decision.explanation,
|
|
2688
|
+
// A1/A2/C3 — carry the capability verdict + override reason so the trace
|
|
2689
|
+
// and console can show WHY the pair was chosen, not just which pair.
|
|
2690
|
+
agenticCapable: decision.agenticCapable,
|
|
2691
|
+
overrideReason: decision.overrideReason,
|
|
2291
2692
|
};
|
|
2292
2693
|
// Walk the ranked candidates (winner first) and return the first available
|
|
2293
2694
|
// provider — never a provider that lacks a key or endpoint. Providers that
|
|
@@ -2446,6 +2847,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
2446
2847
|
provider: candidate.provider,
|
|
2447
2848
|
model,
|
|
2448
2849
|
score: decision.score,
|
|
2850
|
+
agenticCapable: isAgenticCapableModel(model, candidate.provider),
|
|
2851
|
+
overrideReason: decision.overrideReason,
|
|
2852
|
+
...(opts?.fallbackFrom ? { fallbackFrom: opts.fallbackFrom } : {}),
|
|
2449
2853
|
});
|
|
2450
2854
|
return {
|
|
2451
2855
|
type: resolved.type,
|
|
@@ -2454,6 +2858,12 @@ export class ChatCommand extends BaseCommand {
|
|
|
2454
2858
|
ranked: candidates,
|
|
2455
2859
|
complexity: decision.complexity,
|
|
2456
2860
|
score: decision.score,
|
|
2861
|
+
agenticCapable: isAgenticCapableModel(model, resolved.type),
|
|
2862
|
+
overrideReason: decision.overrideReason,
|
|
2863
|
+
taskProfile: {
|
|
2864
|
+
intent: decision.taskProfile.intent,
|
|
2865
|
+
requiresVerification: decision.taskProfile.requiresVerification,
|
|
2866
|
+
},
|
|
2457
2867
|
};
|
|
2458
2868
|
}
|
|
2459
2869
|
}
|
|
@@ -2477,6 +2887,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
2477
2887
|
provider: usableProvider,
|
|
2478
2888
|
model: decision.model,
|
|
2479
2889
|
score: decision.score,
|
|
2890
|
+
agenticCapable: isAgenticCapableModel(decision.model, usableProvider),
|
|
2891
|
+
overrideReason: decision.overrideReason,
|
|
2892
|
+
...(opts?.fallbackFrom ? { fallbackFrom: opts.fallbackFrom } : {}),
|
|
2480
2893
|
});
|
|
2481
2894
|
const resolved = resolveProvider(this.configManager, usableProvider);
|
|
2482
2895
|
const model = (await resolveRoute({
|