agent-nuvira 3.3.10 → 3.3.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -1
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +34 -12
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/cli/chat.d.ts +36 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +440 -44
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/config.d.ts +28 -0
- package/dist/cli/config.d.ts.map +1 -1
- package/dist/cli/config.js +169 -1
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/eval.d.ts +8 -0
- package/dist/cli/eval.d.ts.map +1 -1
- package/dist/cli/eval.js +64 -0
- package/dist/cli/eval.js.map +1 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +8 -0
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/loop-executor.d.ts +13 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +27 -18
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/tool-install-prompt.d.ts +58 -0
- package/dist/cli/tool-install-prompt.d.ts.map +1 -1
- package/dist/cli/tool-install-prompt.js +137 -0
- package/dist/cli/tool-install-prompt.js.map +1 -1
- package/dist/cli/weak-model-prompt.d.ts +16 -7
- package/dist/cli/weak-model-prompt.d.ts.map +1 -1
- package/dist/cli/weak-model-prompt.js +24 -8
- package/dist/cli/weak-model-prompt.js.map +1 -1
- package/dist/config/capability-mode.d.ts +122 -0
- package/dist/config/capability-mode.d.ts.map +1 -0
- package/dist/config/capability-mode.js +132 -0
- package/dist/config/capability-mode.js.map +1 -0
- package/dist/config/limits.d.ts +47 -0
- package/dist/config/limits.d.ts.map +1 -0
- package/dist/config/limits.js +64 -0
- package/dist/config/limits.js.map +1 -0
- package/dist/config/process-env.d.ts.map +1 -1
- package/dist/config/process-env.js +44 -0
- package/dist/config/process-env.js.map +1 -1
- package/dist/config/types.d.ts +113 -0
- package/dist/config/types.d.ts.map +1 -1
- package/dist/config/work-digest.d.ts +46 -0
- package/dist/config/work-digest.d.ts.map +1 -0
- package/dist/config/work-digest.js +80 -0
- package/dist/config/work-digest.js.map +1 -0
- package/dist/context/cache.d.ts +35 -1
- package/dist/context/cache.d.ts.map +1 -1
- package/dist/context/cache.js +31 -2
- package/dist/context/cache.js.map +1 -1
- package/dist/gateway/inbound-media.d.ts +18 -0
- package/dist/gateway/inbound-media.d.ts.map +1 -1
- package/dist/gateway/inbound-media.js +96 -2
- package/dist/gateway/inbound-media.js.map +1 -1
- package/dist/inference/anthropic-adapter.d.ts +6 -0
- package/dist/inference/anthropic-adapter.d.ts.map +1 -1
- package/dist/inference/anthropic-adapter.js +79 -8
- package/dist/inference/anthropic-adapter.js.map +1 -1
- package/dist/inference/gemini-adapter.d.ts +6 -0
- package/dist/inference/gemini-adapter.d.ts.map +1 -1
- package/dist/inference/gemini-adapter.js +73 -8
- package/dist/inference/gemini-adapter.js.map +1 -1
- package/dist/inference/groq-adapter.d.ts +3 -0
- package/dist/inference/groq-adapter.d.ts.map +1 -1
- package/dist/inference/groq-adapter.js +83 -40
- package/dist/inference/groq-adapter.js.map +1 -1
- package/dist/inference/interface.d.ts +7 -0
- package/dist/inference/interface.d.ts.map +1 -1
- package/dist/inference/model-catalog.d.ts +17 -0
- package/dist/inference/model-catalog.d.ts.map +1 -1
- package/dist/inference/model-catalog.js +42 -7
- package/dist/inference/model-catalog.js.map +1 -1
- package/dist/inference/model-probe.d.ts +17 -0
- package/dist/inference/model-probe.d.ts.map +1 -1
- package/dist/inference/model-probe.js +58 -1
- package/dist/inference/model-probe.js.map +1 -1
- package/dist/inference/model-validator.d.ts +16 -1
- package/dist/inference/model-validator.d.ts.map +1 -1
- package/dist/inference/model-validator.js +83 -2
- package/dist/inference/model-validator.js.map +1 -1
- package/dist/inference/nim-adapter.js +4 -4
- package/dist/inference/nim-adapter.js.map +1 -1
- package/dist/inference/openai-compat-adapter.d.ts +9 -0
- package/dist/inference/openai-compat-adapter.d.ts.map +1 -1
- package/dist/inference/openai-compat-adapter.js +101 -51
- package/dist/inference/openai-compat-adapter.js.map +1 -1
- package/dist/inference/openrouter-adapter.d.ts +3 -0
- package/dist/inference/openrouter-adapter.d.ts.map +1 -1
- package/dist/inference/openrouter-adapter.js +134 -57
- package/dist/inference/openrouter-adapter.js.map +1 -1
- package/dist/inference/reasoning-effort.d.ts +130 -0
- package/dist/inference/reasoning-effort.d.ts.map +1 -0
- package/dist/inference/reasoning-effort.js +237 -0
- package/dist/inference/reasoning-effort.js.map +1 -0
- package/dist/inference/route-resolver.d.ts +7 -0
- package/dist/inference/route-resolver.d.ts.map +1 -1
- package/dist/inference/route-resolver.js +2 -1
- package/dist/inference/route-resolver.js.map +1 -1
- package/dist/inference/sse.d.ts +6 -0
- package/dist/inference/sse.d.ts.map +1 -1
- package/dist/inference/sse.js +1 -0
- package/dist/inference/sse.js.map +1 -1
- package/dist/inference/tool-call-utils.d.ts +4 -3
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +16 -4
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/inference/tools.d.ts +13 -0
- package/dist/inference/tools.d.ts.map +1 -1
- package/dist/inference/tools.js +20 -2
- package/dist/inference/tools.js.map +1 -1
- package/dist/learning/agentic-route-gate.d.ts +104 -0
- package/dist/learning/agentic-route-gate.d.ts.map +1 -0
- package/dist/learning/agentic-route-gate.js +125 -0
- package/dist/learning/agentic-route-gate.js.map +1 -0
- package/dist/learning/auto-router.d.ts +36 -0
- package/dist/learning/auto-router.d.ts.map +1 -1
- package/dist/learning/auto-router.js +115 -8
- package/dist/learning/auto-router.js.map +1 -1
- package/dist/learning/autonomy-policy.d.ts +8 -0
- package/dist/learning/autonomy-policy.d.ts.map +1 -1
- package/dist/learning/autonomy-policy.js +34 -0
- package/dist/learning/autonomy-policy.js.map +1 -1
- package/dist/learning/build-prerequisites.d.ts +68 -0
- package/dist/learning/build-prerequisites.d.ts.map +1 -0
- package/dist/learning/build-prerequisites.js +267 -0
- package/dist/learning/build-prerequisites.js.map +1 -0
- package/dist/learning/capability-parity.d.ts +108 -0
- package/dist/learning/capability-parity.d.ts.map +1 -0
- package/dist/learning/capability-parity.js +154 -0
- package/dist/learning/capability-parity.js.map +1 -0
- package/dist/learning/continuation-intent.d.ts +43 -0
- package/dist/learning/continuation-intent.d.ts.map +1 -0
- package/dist/learning/continuation-intent.js +82 -0
- package/dist/learning/continuation-intent.js.map +1 -0
- package/dist/learning/cost-tracker.d.ts +28 -4
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +58 -8
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/eval-framework.d.ts +17 -0
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +63 -2
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/model-capability.d.ts +94 -0
- package/dist/learning/model-capability.d.ts.map +1 -0
- package/dist/learning/model-capability.js +172 -0
- package/dist/learning/model-capability.js.map +1 -0
- package/dist/learning/model-harness.d.ts +19 -0
- package/dist/learning/model-harness.d.ts.map +1 -1
- package/dist/learning/model-harness.js +27 -0
- package/dist/learning/model-harness.js.map +1 -1
- package/dist/learning/model-registry.d.ts +33 -0
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +79 -0
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/prompt-budget.d.ts +91 -0
- package/dist/learning/prompt-budget.d.ts.map +1 -0
- package/dist/learning/prompt-budget.js +99 -0
- package/dist/learning/prompt-budget.js.map +1 -0
- package/dist/learning/reasoning-trace.d.ts +55 -2
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +48 -0
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +16 -3
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/resolve-options.d.ts.map +1 -1
- package/dist/learning/resolve-options.js +16 -3
- package/dist/learning/resolve-options.js.map +1 -1
- package/dist/learning/routing-history.d.ts +13 -0
- package/dist/learning/routing-history.d.ts.map +1 -1
- package/dist/learning/routing-history.js.map +1 -1
- package/dist/learning/run-trace.d.ts +68 -0
- package/dist/learning/run-trace.d.ts.map +1 -1
- package/dist/learning/run-trace.js +101 -0
- package/dist/learning/run-trace.js.map +1 -1
- package/dist/learning/turn-report.d.ts +78 -0
- package/dist/learning/turn-report.d.ts.map +1 -0
- package/dist/learning/turn-report.js +140 -0
- package/dist/learning/turn-report.js.map +1 -0
- package/dist/tools/edit-verification.d.ts +2 -11
- package/dist/tools/edit-verification.d.ts.map +1 -1
- package/dist/tools/edit-verification.js +33 -1
- package/dist/tools/edit-verification.js.map +1 -1
- package/dist/tools/loop-skill-hint.d.ts +78 -0
- package/dist/tools/loop-skill-hint.d.ts.map +1 -1
- package/dist/tools/loop-skill-hint.js +274 -5
- package/dist/tools/loop-skill-hint.js.map +1 -1
- package/dist/tools/plan-store.d.ts +104 -3
- package/dist/tools/plan-store.d.ts.map +1 -1
- package/dist/tools/plan-store.js +250 -6
- package/dist/tools/plan-store.js.map +1 -1
- package/dist/tools/read-extract.d.ts +9 -3
- package/dist/tools/read-extract.d.ts.map +1 -1
- package/dist/tools/read-extract.js +70 -25
- package/dist/tools/read-extract.js.map +1 -1
- package/dist/tools/registry.d.ts +20 -1
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +17 -4
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +34 -0
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +149 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +706 -13
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/utils/effect-verification.js +1 -1
- package/dist/utils/effect-verification.js.map +1 -1
- package/dist/web-dashboard/attachment-extract.d.ts +8 -1
- package/dist/web-dashboard/attachment-extract.d.ts.map +1 -1
- package/dist/web-dashboard/attachment-extract.js +14 -3
- package/dist/web-dashboard/attachment-extract.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +21 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +33 -5
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +198 -21
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +97 -0
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.js +44 -4
- package/dist/web-dashboard/workspace-guard.js.map +1 -1
- package/package.json +4 -1
- package/src/web-dashboard/public/assets/index-BrZcYhy6.js +207 -0
- package/src/web-dashboard/public/assets/index-BrZcYhy6.js.map +1 -0
- package/src/web-dashboard/public/assets/index-DsWczTa6.css +1 -0
- package/src/web-dashboard/public/index.html +2 -2
- package/src/web-dashboard/public/assets/index-Beportyl.js +0 -207
- package/src/web-dashboard/public/assets/index-Beportyl.js.map +0 -1
- package/src/web-dashboard/public/assets/index-gtQyg9nm.css +0 -1
package/dist/cli/chat.js
CHANGED
|
@@ -22,10 +22,12 @@ import { printOrchestrationResult } from './execute.js';
|
|
|
22
22
|
import { beginIsolation, endIsolation, resolveIsolationRequest, worktreeNotice, } from '../tools/worktree.js';
|
|
23
23
|
import { closeResume, openResume, resolveResumeRequest, } from '../learning/step-checkpoint.js';
|
|
24
24
|
import { applyActiveModel } from './model.js';
|
|
25
|
+
import { capabilityReasoningEffort, isMaxCapability } from '../config/capability-mode.js';
|
|
25
26
|
import { getProviderFallback, classifyFallbackError, isRetryableError, isTransientForRetry, recordRegistrySuccess } from '../learning/provider-fallback.js';
|
|
26
27
|
import { recordActionFailure } from '../learning/failure-bookkeeping.js';
|
|
27
28
|
import { resolveThreadBudgetChars } from '../learning/context-budget.js';
|
|
28
29
|
import { getAutoRouter, isAutoModel, isAutoProvider, governanceVerdict } from '../learning/auto-router.js';
|
|
30
|
+
import { continuationSoftwareText } from '../learning/continuation-intent.js';
|
|
29
31
|
import { estimateTokens } from '../learning/cost-tracker.js';
|
|
30
32
|
import { getModelRegistry } from '../learning/model-registry.js';
|
|
31
33
|
import { refreshModelRegistry } from '../inference/model-probe.js';
|
|
@@ -33,9 +35,9 @@ import { startWarmupDaemon } from '../learning/model-warmup.js';
|
|
|
33
35
|
import { recordRoutingDecision } from '../learning/routing-history.js';
|
|
34
36
|
import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
|
|
35
37
|
import { buildAutoResolveOptions } from '../learning/resolve-options.js';
|
|
36
|
-
import { buildDeepFailoverPool, createFailoverExclusionFilter } from '../learning/resilient-call.js';
|
|
38
|
+
import { buildDeepFailoverPool, createFailoverExclusionFilter, modelBreadthReport } from '../learning/resilient-call.js';
|
|
37
39
|
import { parseRequestSync } from '../nlu/parser.js';
|
|
38
|
-
import {
|
|
40
|
+
import { createPersistentPlanStore } from '../tools/plan-store.js';
|
|
39
41
|
import { withLogCorrelation } from '../enterprise/log.js';
|
|
40
42
|
import { recordMetricTime, getMetrics } from '../enterprise/metrics.js';
|
|
41
43
|
import { resolveDispatch } from '../nlu/actions.js';
|
|
@@ -44,10 +46,13 @@ import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
|
|
|
44
46
|
// WS3 (#25) — the turn as a span, when an operator has asked for OTLP export.
|
|
45
47
|
import { flushSpans, otelNoticeOnce, startTurnSpan } from '../observability/otel.js';
|
|
46
48
|
import { detectAnswerQualityFailure, answerQualityError, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
|
|
47
|
-
import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, buildTraceOutcome, traceOutcomeSucceeded } from '../learning/reasoning-trace.js';
|
|
49
|
+
import { beginTrace, endTrace, recordStep, recordTraceEvent, recordTraceFindings, recordTurnReport, buildTraceOutcome, traceOutcomeSucceeded } from '../learning/reasoning-trace.js';
|
|
48
50
|
import { recordWorkingState, getWorkingState, formatWorkingState, isProjectLedgerDir } from '../learning/working-state.js';
|
|
49
51
|
import { getLoopExposureMode } from '../tools/toolsets.js';
|
|
50
|
-
import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
|
|
52
|
+
import { resolveModelHarnessProfile, shouldSkipNativeTools, isAgenticCapableModel } from '../learning/model-harness.js';
|
|
53
|
+
import { assertAgenticRoute, setWeakModelConsent, resolveWeakModelPolicy, weakRouteNotice } from '../learning/agentic-route-gate.js';
|
|
54
|
+
import { resolvePromptBudget, measurePromptBudget, formatPromptBudgetBreakdown } from '../learning/prompt-budget.js';
|
|
55
|
+
import { buildTurnReport, formatTurnReport } from '../learning/turn-report.js';
|
|
51
56
|
import { resolveAdapterDefault, hasCredentials } from '../learning/model-selection.js';
|
|
52
57
|
import { buildLoopProjectContext } from '../tools/loop-project-context.js';
|
|
53
58
|
import { sweepTransientFailures, collectionRevivalStore } from '../learning/provider-revival.js';
|
|
@@ -280,7 +285,12 @@ export async function generateWithTransientRetry(attempt, signal, onRetry) {
|
|
|
280
285
|
}
|
|
281
286
|
}
|
|
282
287
|
}
|
|
283
|
-
|
|
288
|
+
/**
|
|
289
|
+
* Exported for the release gate (`tests/release/agent-contracts.test.ts`): the
|
|
290
|
+
* assembled system prompt's SIZE and its required contract clauses are a
|
|
291
|
+
* release invariant, not an implementation detail. See the gate for why.
|
|
292
|
+
*/
|
|
293
|
+
export function buildToolSystemPrompt(parsed) {
|
|
284
294
|
return [
|
|
285
295
|
"You are Nuvira, Agent-Nuvira's AI agent. You code, create, write, analyze, and automate — anything the user needs. You identify as Nuvira (never 'Buff').",
|
|
286
296
|
'Be precise and honest. When a request is ambiguous or incomplete, clarify with ask_user instead of guessing.',
|
|
@@ -304,9 +314,54 @@ function buildToolSystemPrompt(parsed) {
|
|
|
304
314
|
'- Some tools live OUTSIDE your visible list in domain toolsets (media, browser, channels, docker, …). If a tool you need is "unknown", call tool_search with {"action":"load","toolset":"<name>"} — its tools become callable immediately.',
|
|
305
315
|
'- Always end with suggest_followups.',
|
|
306
316
|
'',
|
|
317
|
+
// WHY THIS BLOCK EXISTS. A weak model, primed by an earlier "attach a
|
|
318
|
+
// project folder" refusal, carried that refusal into a plain writing
|
|
319
|
+
// request: asked to write an essay it answered with instructions for moving
|
|
320
|
+
// a project folder and "I don't have the capability to create files in a
|
|
321
|
+
// workspace". Both are wrong and both are exactly what these lines forbid.
|
|
322
|
+
'## Workspace and general requests',
|
|
323
|
+
'- Writing and questions do NOT need a project folder. An essay, poem, email, explanation, summary or brainstorm is answered directly in your reply — never mention folders, directories, "attaching", or the workspace for these.',
|
|
324
|
+
'- You CAN create and edit files (write_file) and run commands (run_terminal) when the request needs them. Never say you cannot create or write files.',
|
|
325
|
+
'- Never ask the user to attach a folder, and never tell them to move files into a directory. If a request genuinely needs a workspace and none is attached, the app asks for one on its own — so just answer as best you can.',
|
|
326
|
+
'- Treat a request as project work ONLY when it actually concerns the codebase or produces a file on disk. A general question that happens to contain a word like "class", "project", "test" or "document" is still a general question.',
|
|
327
|
+
'',
|
|
328
|
+
// WHY THIS BLOCK EXISTS. Asked to build a Tauri app, a model ran
|
|
329
|
+
// `cargo --version` → exit 127 and then answered "I cannot install
|
|
330
|
+
// system-level software like Rust on your host machine — I am physically
|
|
331
|
+
// unable to build the Tauri application", handing the user a manual
|
|
332
|
+
// `curl … | sh` step. It was not sandboxed at all; it had a real shell and
|
|
333
|
+
// the user had just granted terminal permission (trace
|
|
334
|
+
// trace-1791127992452-qzgodi). A missing prerequisite is a step to
|
|
335
|
+
// complete, not a capability limit — so this states that plainly.
|
|
336
|
+
'## Toolchains and missing prerequisites',
|
|
337
|
+
'- You run on the user\u2019s REAL machine with a REAL shell (run_terminal). You are NOT sandboxed and NOT "enclosed".',
|
|
338
|
+
'- If a command fails because a tool is missing (`command not found`, exit 127), INSTALL it yourself with run_terminal — e.g. `brew install <tool>`, `xcode-select --install`, the language installer (rustup for cargo/rust, nvm/npm for node, pyenv/pip for python, go, …), or apt/winget — then re-run the original command.',
|
|
339
|
+
'- NEVER answer that you are unable or not permitted to install a prerequisite, and never hand the user an install step you were asked to do. A missing tool is a step to complete, not a wall.',
|
|
340
|
+
'',
|
|
307
341
|
TOOL_CONTRACT_JSON,
|
|
308
342
|
].join('\n');
|
|
309
343
|
}
|
|
344
|
+
/**
|
|
345
|
+
* #30 — does this turn carry an HONESTY FLAG?
|
|
346
|
+
*
|
|
347
|
+
* A flagged answer must NEVER be written to the shared response cache. The cache
|
|
348
|
+
* stores text only, so a later identical prompt — on the CLI, the dashboard or the
|
|
349
|
+
* gateway — was served the reply with every flag absent: a known-unverified claim
|
|
350
|
+
* (`unverifiedActionClaim` / `unverifiedEditClaim` / `unverifiedBuildClaim`), an
|
|
351
|
+
* announced-but-unperformed action (`unfulfilledPromise`), a missing deliverable
|
|
352
|
+
* (`undeliveredArtifact`) or an inert turn (`noActionTaken`) replayed as a clean
|
|
353
|
+
* answer, on every surface at once. The flags exist because those answers must
|
|
354
|
+
* not be replayed as settled, so the turn is not cached and re-derives instead.
|
|
355
|
+
*/
|
|
356
|
+
export function turnCarriesHonestyFlag(result) {
|
|
357
|
+
return Boolean(result.unverifiedActionClaim ||
|
|
358
|
+
result.unfulfilledPromise ||
|
|
359
|
+
result.undeliveredArtifact ||
|
|
360
|
+
result.unverifiedBuildClaim ||
|
|
361
|
+
result.unverifiedEdit ||
|
|
362
|
+
result.unverifiedEditClaim ||
|
|
363
|
+
result.noActionTaken);
|
|
364
|
+
}
|
|
310
365
|
// ─── ChatCommand ────────────────────────────────────────────────────────────
|
|
311
366
|
export class ChatCommand extends BaseCommand {
|
|
312
367
|
devModeAuto = false;
|
|
@@ -352,7 +407,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
352
407
|
* console injects a per-session store instead; this is the CLI/execute
|
|
353
408
|
* default so a plan survives across turns within one chat session).
|
|
354
409
|
*/
|
|
355
|
-
planStore =
|
|
410
|
+
planStore = createPersistentPlanStore(`cli:${process.cwd()}`);
|
|
356
411
|
/**
|
|
357
412
|
* Whether the cold-start probe has fired this session. On a fresh registry
|
|
358
413
|
* (no verified models yet) the FIRST auto pick fires a background
|
|
@@ -428,11 +483,27 @@ export class ChatCommand extends BaseCommand {
|
|
|
428
483
|
? await this.getProvider({})
|
|
429
484
|
: await this.getProvider(mergedOpts);
|
|
430
485
|
let model = mergedOpts.model;
|
|
486
|
+
// B — captured from the route so the post-route capability gate below can
|
|
487
|
+
// judge the FINAL pair without re-resolving anything.
|
|
488
|
+
let routeVerdict;
|
|
489
|
+
let routedText;
|
|
431
490
|
if (autoMode) {
|
|
432
|
-
|
|
491
|
+
// Intent-aware escalation: a bare "yes"/"do it" continuing software work
|
|
492
|
+
// routes on the prior ask, not on the signal-free continuation. When
|
|
493
|
+
// there is no such hint the call is byte-identical to before.
|
|
494
|
+
const routingText = continuationSoftwareText(message, opts.history ?? []) ?? undefined;
|
|
495
|
+
const routed = routingText
|
|
496
|
+
? await this.routeMessageAuto(message, [], { routingText })
|
|
497
|
+
: await this.routeMessageAuto(message);
|
|
433
498
|
type = routed.type;
|
|
434
499
|
provider = routed.provider;
|
|
435
500
|
model = routed.model;
|
|
501
|
+
routedText = routingText;
|
|
502
|
+
routeVerdict = {
|
|
503
|
+
complexity: routed.complexity,
|
|
504
|
+
agenticCapable: routed.agenticCapable,
|
|
505
|
+
taskProfile: routed.taskProfile,
|
|
506
|
+
};
|
|
436
507
|
}
|
|
437
508
|
// P3 — tell the GUI where the turn is headed before the tool loop runs.
|
|
438
509
|
const isLocalFallback = autoMode && type === 'local';
|
|
@@ -440,6 +511,85 @@ export class ChatCommand extends BaseCommand {
|
|
|
440
511
|
? ' ⚠️ local model only — run `nuvira models` or `nuvira provider set` to add a cloud provider'
|
|
441
512
|
: '';
|
|
442
513
|
opts.onProgress?.(` 🧠 routed to ${provider.name}${model ? ` / ${model}` : ''} — working…${localWarning}`);
|
|
514
|
+
// ── Workstream B — agentic capability gate (consent-first) ──────────────
|
|
515
|
+
// The router RECORDED a verdict (A1); the turn must ACT on it. A software/
|
|
516
|
+
// agentic ask must never silently run on a weak model. Consent is per
|
|
517
|
+
// SESSION: one answer covers this session, and a new chat/task asks again.
|
|
518
|
+
//
|
|
519
|
+
// Interactive = an injected askUser (the dashboard console) or a real TTY.
|
|
520
|
+
// A piped/headless run never reaches the ask — it falls to the configured
|
|
521
|
+
// policy, whose default (`deny`/retry) can never silently downgrade.
|
|
522
|
+
if (autoMode && routeVerdict && routeVerdict.agenticCapable === false) {
|
|
523
|
+
const sessionId = opts.debugSession;
|
|
524
|
+
const interactive = Boolean(opts.askUser) || Boolean(process.stdin.isTTY);
|
|
525
|
+
const policy = resolveWeakModelPolicy(this.configManager);
|
|
526
|
+
const asDecision = {
|
|
527
|
+
complexity: routeVerdict.complexity,
|
|
528
|
+
taskProfile: (routeVerdict.taskProfile ?? {
|
|
529
|
+
intent: 'unknown',
|
|
530
|
+
requiresVerification: false,
|
|
531
|
+
}),
|
|
532
|
+
provider: type,
|
|
533
|
+
model: model ?? '',
|
|
534
|
+
agenticCapable: false,
|
|
535
|
+
};
|
|
536
|
+
let gate = assertAgenticRoute(asDecision, { sessionId, policy, interactive });
|
|
537
|
+
if (gate.action === 'ask') {
|
|
538
|
+
try {
|
|
539
|
+
const ask = opts.askUser ??
|
|
540
|
+
(await import('../tools/ask-user.js')).renderAskUser;
|
|
541
|
+
const answer = await ask(`This is a software task, but only a weak model is available ` +
|
|
542
|
+
`(${provider.name}${model ? ` / ${model}` : ''}). How should I proceed for this session?`, [
|
|
543
|
+
{ label: 'Approve the weak model for this session' },
|
|
544
|
+
{ label: 'Wait for a strong model only' },
|
|
545
|
+
], false);
|
|
546
|
+
const granted = Number(answer.index) === 0;
|
|
547
|
+
if (sessionId)
|
|
548
|
+
setWeakModelConsent(sessionId, granted ? 'granted' : 'denied');
|
|
549
|
+
gate = { ...gate, action: granted ? 'proceed-weak-consented' : 'retry-strong' };
|
|
550
|
+
}
|
|
551
|
+
catch {
|
|
552
|
+
// No answer reachable — treat as deny (never a silent downgrade).
|
|
553
|
+
gate = { ...gate, action: 'retry-strong' };
|
|
554
|
+
}
|
|
555
|
+
}
|
|
556
|
+
if (gate.action === 'retry-strong') {
|
|
557
|
+
// Try ONE re-route that EXCLUDES the weak provider; accept it only if
|
|
558
|
+
// the returned pair is genuinely agentic-capable. Nothing capable →
|
|
559
|
+
// refuse honestly rather than run the weak model against consent.
|
|
560
|
+
let next = null;
|
|
561
|
+
try {
|
|
562
|
+
next = routedText
|
|
563
|
+
? await this.routeMessageAuto(message, [type], { routingText: routedText })
|
|
564
|
+
: await this.routeMessageAuto(message, [type]);
|
|
565
|
+
}
|
|
566
|
+
catch {
|
|
567
|
+
next = null;
|
|
568
|
+
}
|
|
569
|
+
if (next && isAgenticCapableModel(next.model, next.type)) {
|
|
570
|
+
type = next.type;
|
|
571
|
+
provider = next.provider;
|
|
572
|
+
model = next.model;
|
|
573
|
+
opts.onProgress?.(` 🧠 re-routed to an agentic-capable model: ${provider.name}${model ? ` / ${model}` : ''}`);
|
|
574
|
+
}
|
|
575
|
+
else {
|
|
576
|
+
return {
|
|
577
|
+
content: gate.notice ??
|
|
578
|
+
'No agentic-capable model is available for this software task right now.',
|
|
579
|
+
followups: [],
|
|
580
|
+
generationFailed: true,
|
|
581
|
+
refused: true,
|
|
582
|
+
provider: type,
|
|
583
|
+
model,
|
|
584
|
+
transport: 'none',
|
|
585
|
+
};
|
|
586
|
+
}
|
|
587
|
+
}
|
|
588
|
+
else if (gate.notice) {
|
|
589
|
+
// proceed-weak-consented — say so plainly; never a silent weak model.
|
|
590
|
+
opts.onProgress?.(` ${gate.notice}`);
|
|
591
|
+
}
|
|
592
|
+
}
|
|
443
593
|
// P4 — when a project is attached, recall its prior sessions + facts
|
|
444
594
|
// FRESH per turn (the snapshot is cached, the recall is not — prior work
|
|
445
595
|
// may have landed since the last turn). Best-effort: empty recall injects
|
|
@@ -501,11 +651,48 @@ export class ChatCommand extends BaseCommand {
|
|
|
501
651
|
...this.turnEnvelopeOf(answer),
|
|
502
652
|
};
|
|
503
653
|
}
|
|
654
|
+
// E1 — assemble the turn report from RECORDED evidence (plan store, tool
|
|
655
|
+
// outcomes, honesty flags). Derived, never narrated, so the trust verdict
|
|
656
|
+
// cannot be talked up by the model.
|
|
657
|
+
let turnReport;
|
|
658
|
+
try {
|
|
659
|
+
const planSnapshot = (opts.planStore ?? this.planStore).snapshot?.() ?? null;
|
|
660
|
+
turnReport = buildTurnReport({
|
|
661
|
+
goal: message,
|
|
662
|
+
plan: planSnapshot,
|
|
663
|
+
toolCalls: answer.toolCalls,
|
|
664
|
+
successfulToolCalls: answer.successfulToolCalls,
|
|
665
|
+
mutations: answer.runTrace?.mutations,
|
|
666
|
+
changedPaths: answer.runTrace?.paths,
|
|
667
|
+
flags: {
|
|
668
|
+
unverifiedActionClaim: answer.unverifiedActionClaim,
|
|
669
|
+
unverifiedEdit: answer.unverifiedEdit,
|
|
670
|
+
unverifiedEditClaim: answer.unverifiedEditClaim,
|
|
671
|
+
unverifiedBuildClaim: answer.unverifiedBuildClaim,
|
|
672
|
+
undeliveredArtifact: answer.undeliveredArtifact,
|
|
673
|
+
unfulfilledPromise: answer.unfulfilledPromise,
|
|
674
|
+
noActionTaken: answer.noActionTaken,
|
|
675
|
+
},
|
|
676
|
+
});
|
|
677
|
+
}
|
|
678
|
+
catch {
|
|
679
|
+
// A report must never break the turn.
|
|
680
|
+
}
|
|
681
|
+
// E-trace — persist the report on the turn's reasoning trace so it is
|
|
682
|
+
// reviewable after the fact (the Trace tab renders it), not only in this
|
|
683
|
+
// turn's return value. Best-effort: a trace write never breaks a turn.
|
|
684
|
+
recordTurnReport(answer.traceId, turnReport);
|
|
685
|
+
// E3 — surface a non-trivial report on the console. A plain answer (no
|
|
686
|
+
// plan, nothing changed) produces no summary and stays silent.
|
|
687
|
+
if (turnReport?.summary) {
|
|
688
|
+
opts.onProgress?.(formatTurnReport(turnReport));
|
|
689
|
+
}
|
|
504
690
|
// E3b: strip raw suggest_followups JSON embedded in content by the model
|
|
505
691
|
const cleanContent = stripToolCallArtifacts(answer.content || '');
|
|
506
692
|
return {
|
|
507
693
|
content: cleanContent,
|
|
508
694
|
followups: answer.followups ?? [],
|
|
695
|
+
...(turnReport ? { turnReport } : {}),
|
|
509
696
|
generationFailed: answer.generationFailed,
|
|
510
697
|
cancelled: answer.cancelled,
|
|
511
698
|
bounded: answer.bounded,
|
|
@@ -800,7 +987,12 @@ export class ChatCommand extends BaseCommand {
|
|
|
800
987
|
// so context-fit routing reacts to a long session, not just the task
|
|
801
988
|
// text. Estimation only — never a hard block.
|
|
802
989
|
const historyEstimate = estimateTokens(history.map((h) => h.content).join('\n') + '\n' + message);
|
|
803
|
-
|
|
990
|
+
// Intent-aware escalation for a bare continuation of software work.
|
|
991
|
+
const routingText = continuationSoftwareText(message, history) ?? undefined;
|
|
992
|
+
const routed = await this.routeMessageAuto(message, [], {
|
|
993
|
+
contextHintTokens: historyEstimate,
|
|
994
|
+
...(routingText ? { routingText } : {}),
|
|
995
|
+
});
|
|
804
996
|
type = routed.type;
|
|
805
997
|
provider = routed.provider;
|
|
806
998
|
effectiveModel = routed.model;
|
|
@@ -962,25 +1154,35 @@ export class ChatCommand extends BaseCommand {
|
|
|
962
1154
|
const cacheModel = this.cacheModelFor(session);
|
|
963
1155
|
if (cacheEnabled) {
|
|
964
1156
|
try {
|
|
965
|
-
|
|
966
|
-
|
|
1157
|
+
// #30 — read the entry WITH its recorded activity, not just the text: a
|
|
1158
|
+
// replay that dropped `toolCalls` rendered no tool cards on the dashboard
|
|
1159
|
+
// while the first run did, so a repeated prompt read as a turn that did
|
|
1160
|
+
// nothing. The text alone is still what the answer is; the activity is
|
|
1161
|
+
// reported so the surface is honest about what the cached turn DID.
|
|
1162
|
+
const cached = await cache.getEntry(message, cacheModel, session.type, turnScope);
|
|
1163
|
+
if (cached) {
|
|
967
1164
|
// NOTE: the cached answer is NOT printed here — the caller prints
|
|
968
1165
|
// content AFTER runChatAnswer returns (answer-first ordering). A
|
|
969
1166
|
// print here would show the answer before the turn's own progress
|
|
970
1167
|
// lines AND double-print it.
|
|
971
1168
|
history.push({ role: 'user', content: message });
|
|
972
|
-
history.push({ role: 'assistant', content:
|
|
973
|
-
this.memoryNoteTurn(message,
|
|
1169
|
+
history.push({ role: 'assistant', content: cached.response });
|
|
1170
|
+
this.memoryNoteTurn(message, cached.response);
|
|
974
1171
|
// WS2 — a cache replay reached no model, so the log says exactly that
|
|
975
1172
|
// rather than borrowing an attribution from a turn that did not run.
|
|
976
1173
|
// The workspace the replayed answer belongs to is recorded with the hit:
|
|
977
1174
|
// a cache replay does no work, so "which project is this answer about?"
|
|
978
1175
|
// is the one fact needed to tell a replay from a real turn.
|
|
979
|
-
debugLog?.event('cache.hit', { chars:
|
|
1176
|
+
debugLog?.event('cache.hit', { chars: cached.response.length, scope: turnScope });
|
|
980
1177
|
const cacheNotice = debugLogNotice(ctxOverrides?.debugSurface ?? 'cli-chat', debugLog?.write() ?? null);
|
|
981
1178
|
if (cacheNotice)
|
|
982
1179
|
logger.info(cacheNotice);
|
|
983
|
-
return {
|
|
1180
|
+
return {
|
|
1181
|
+
content: cached.response,
|
|
1182
|
+
...(cached.toolCalls ? { toolCalls: cached.toolCalls } : {}),
|
|
1183
|
+
...(cached.successfulToolCalls ? { successfulToolCalls: cached.successfulToolCalls } : {}),
|
|
1184
|
+
...(cached.bounded ? { bounded: cached.bounded } : {}),
|
|
1185
|
+
};
|
|
984
1186
|
}
|
|
985
1187
|
}
|
|
986
1188
|
catch {
|
|
@@ -1100,21 +1302,23 @@ export class ChatCommand extends BaseCommand {
|
|
|
1100
1302
|
// computed for the generation-FAILED fallback only — it does NOT reach the
|
|
1101
1303
|
// prompt, so on this surface the MODEL decides and the rules are invisible.
|
|
1102
1304
|
const systemText = buildToolSystemPrompt(parsed);
|
|
1103
|
-
//
|
|
1104
|
-
//
|
|
1105
|
-
//
|
|
1106
|
-
//
|
|
1107
|
-
// (
|
|
1108
|
-
//
|
|
1305
|
+
// Skill hint — MODE-DEPENDENT (see resolveSkillHintMode):
|
|
1306
|
+
// - `match` (default) — the small keyword-matched hint: ONE skill, and
|
|
1307
|
+
// only when the goal really matches; otherwise nothing. This is the
|
|
1308
|
+
// 3.3.10 behaviour and keeps the prompt small.
|
|
1309
|
+
// - `catalog` (opt-in) — the full name+description catalog, which lets the
|
|
1310
|
+
// MODEL pick a skill but costs ~24K chars on every turn, so it must be
|
|
1311
|
+
// chosen (`NUVIRA_SKILL_CATALOG=catalog` or `skills.catalogHint`).
|
|
1312
|
+
// - `off` — never inject one.
|
|
1313
|
+
// Best-effort: any failure returns '' and the turn proceeds byte-identically.
|
|
1109
1314
|
let skillHint = '';
|
|
1110
1315
|
try {
|
|
1111
|
-
const {
|
|
1112
|
-
const injected = {
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
}
|
|
1316
|
+
const { buildConfiguredSkillHint, markLoopSkillUsed } = await import('../tools/loop-skill-hint.js');
|
|
1317
|
+
const injected = {
|
|
1318
|
+
value: null,
|
|
1319
|
+
};
|
|
1320
|
+
skillHint = await buildConfiguredSkillHint(message, this.configManager, injected);
|
|
1321
|
+
await markLoopSkillUsed(injected.value);
|
|
1118
1322
|
}
|
|
1119
1323
|
catch {
|
|
1120
1324
|
skillHint = ''; // best-effort — a hint failure never breaks the turn
|
|
@@ -1273,8 +1477,23 @@ export class ChatCommand extends BaseCommand {
|
|
|
1273
1477
|
}
|
|
1274
1478
|
}
|
|
1275
1479
|
// P0.7 — forward plan mutations to the GUI (structured checklist).
|
|
1276
|
-
if (
|
|
1277
|
-
|
|
1480
|
+
if (event === 'plan:changed') {
|
|
1481
|
+
const snapshot = data;
|
|
1482
|
+
if (ctxOverrides?.onPlanChange) {
|
|
1483
|
+
ctxOverrides.onPlanChange(snapshot);
|
|
1484
|
+
}
|
|
1485
|
+
else {
|
|
1486
|
+
// No GUI consumer (the interactive CLI): show the PROGRESS TABLE in
|
|
1487
|
+
// the terminal so a user watching the run sees the plan advance,
|
|
1488
|
+
// not just the model's narration. A settled plan prints its
|
|
1489
|
+
// achieved SUMMARY instead — the same text the turn closes on.
|
|
1490
|
+
const store = ctxOverrides?.planStore ?? this.planStore;
|
|
1491
|
+
const settled = snapshot.steps.length > 0 &&
|
|
1492
|
+
snapshot.steps.every((s) => s.status === 'done' || s.status === 'blocked');
|
|
1493
|
+
const rendered = settled ? store.summary?.() : store.toTable?.();
|
|
1494
|
+
if (rendered)
|
|
1495
|
+
logger.info(`\n${rendered}\n`);
|
|
1496
|
+
}
|
|
1278
1497
|
}
|
|
1279
1498
|
// P3b — forward git diff payloads to the GUI (the diff card).
|
|
1280
1499
|
if (ctxOverrides?.onGitDiff && event === 'git:diff') {
|
|
@@ -1324,6 +1543,13 @@ export class ChatCommand extends BaseCommand {
|
|
|
1324
1543
|
// default interactive renderer.
|
|
1325
1544
|
...(ctxOverrides?.askUser ? { askUser: ctxOverrides.askUser } : {}),
|
|
1326
1545
|
...(ctxOverrides?.gateway ? { gateway: ctxOverrides.gateway } : {}),
|
|
1546
|
+
// P0.7 — the plan store. This was DROPPED here: `answerOnce` put a
|
|
1547
|
+
// per-session store on `ctxOverrides.planStore`, but only askUser/gateway
|
|
1548
|
+
// were threaded into the tool context, so every surface fell back to the
|
|
1549
|
+
// shared module store — which is why plans leaked across sessions and a
|
|
1550
|
+
// reload started a blank checklist. Thread it (per-session when injected,
|
|
1551
|
+
// else this command's project-scoped store).
|
|
1552
|
+
planStore: ctxOverrides?.planStore ?? this.planStore,
|
|
1327
1553
|
// C2 verify with the actual session model (verify_requirement tool).
|
|
1328
1554
|
callLLM: (prompt, opts) => session.provider.generate(prompt, { ...opts, model: session.model }),
|
|
1329
1555
|
// I3: tools that return {artifact, result} deliverables are recorded to
|
|
@@ -1341,7 +1567,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
1341
1567
|
},
|
|
1342
1568
|
},
|
|
1343
1569
|
};
|
|
1344
|
-
const callModel = this.buildToolCallModel(message, session, options, mode, ctxOverrides?.onToken, ctxOverrides?.signal);
|
|
1570
|
+
const callModel = this.buildToolCallModel(message, session, options, mode, ctxOverrides?.onToken, ctxOverrides?.signal, continuationSoftwareText(message, history) ?? undefined);
|
|
1345
1571
|
let result;
|
|
1346
1572
|
// v1.8x audit — CHAT TRACE CAPTURE: every LLM call in a chat turn is now
|
|
1347
1573
|
// recorded to ~/.nuvira/memory/reasoning-traces.json (source 'chat'), so
|
|
@@ -1359,6 +1585,64 @@ export class ChatCommand extends BaseCommand {
|
|
|
1359
1585
|
});
|
|
1360
1586
|
// G18 — the tool-context emit (declared above) now has somewhere to write.
|
|
1361
1587
|
traceIdForEvents = chatTraceId;
|
|
1588
|
+
// A2 — record the routing DECISION as a first-class event, so a turn that
|
|
1589
|
+
// ran on a weak/incapable model is self-evident in the Trace tab. The
|
|
1590
|
+
// failed Tauri turn had no such record; this is the instrument that would
|
|
1591
|
+
// have shown `agenticCapable:false` at the moment of the choice.
|
|
1592
|
+
if (this.lastRouteSnapshot) {
|
|
1593
|
+
const snap = this.lastRouteSnapshot;
|
|
1594
|
+
recordTraceEvent(chatTraceId, {
|
|
1595
|
+
kind: 'decision',
|
|
1596
|
+
gate: 'routing',
|
|
1597
|
+
summary: `routed to ${snap.provider}/${snap.model} ` +
|
|
1598
|
+
`(complexity ${snap.complexity}, score ${snap.score.toFixed(3)})` +
|
|
1599
|
+
(snap.agenticCapable === false
|
|
1600
|
+
? ` — NOT agentic-capable${snap.overrideReason ? ` (${snap.overrideReason})` : ''}`
|
|
1601
|
+
: ''),
|
|
1602
|
+
routing: snap,
|
|
1603
|
+
});
|
|
1604
|
+
}
|
|
1605
|
+
// D1 — measure the OUTBOUND context and, past the ceiling, degrade the
|
|
1606
|
+
// lowest-value optional contributor first (skill hint → work digest →
|
|
1607
|
+
// recall → …). The identity/tool contract is never trimmed. This is the
|
|
1608
|
+
// guard the 3.3.11 bloat (7.6K → 32.8K chars) never had.
|
|
1609
|
+
try {
|
|
1610
|
+
const contextBudget = resolvePromptBudget(this.configManager);
|
|
1611
|
+
const historyChars = history.reduce((n, h) => n + (h.content?.length ?? 0), 0);
|
|
1612
|
+
const report = measurePromptBudget([
|
|
1613
|
+
{ name: 'system:identity+tool-contract', chars: systemText.length },
|
|
1614
|
+
{ name: 'system:channel-policy', chars: systemPolicyBlock.length },
|
|
1615
|
+
{ name: 'skill-hint', chars: skillHint.length, dropPriority: 10 },
|
|
1616
|
+
{ name: 'working-state', chars: workingStateBlock.length, dropPriority: 25 },
|
|
1617
|
+
{ name: 'recall', chars: ctxOverrides?.recallContext?.length ?? 0, dropPriority: 30 },
|
|
1618
|
+
{
|
|
1619
|
+
name: 'project-context',
|
|
1620
|
+
chars: ctxOverrides?.projectContext?.length ?? ambientProjectContext?.length ?? 0,
|
|
1621
|
+
dropPriority: 40,
|
|
1622
|
+
},
|
|
1623
|
+
{ name: 'file-context', chars: fileContext?.length ?? 0, dropPriority: 45 },
|
|
1624
|
+
{ name: 'history+ask', chars: historyChars },
|
|
1625
|
+
], { budget: contextBudget });
|
|
1626
|
+
// Apply the FIRST ladder step (the documented, safe one): drop the skill
|
|
1627
|
+
// hint from the assembled system message.
|
|
1628
|
+
if (report.trims.includes('skill-hint') && skillHint) {
|
|
1629
|
+
thread[0] = { role: 'system', content: systemText + systemPolicyBlock };
|
|
1630
|
+
}
|
|
1631
|
+
if (report.level !== 'ok') {
|
|
1632
|
+
recordTraceEvent(chatTraceId, {
|
|
1633
|
+
kind: 'decision',
|
|
1634
|
+
gate: 'context-budget',
|
|
1635
|
+
summary: formatPromptBudgetBreakdown(report) +
|
|
1636
|
+
(report.trims.length ? ` — trimmed: ${report.trims.join(', ')}` : ''),
|
|
1637
|
+
});
|
|
1638
|
+
}
|
|
1639
|
+
}
|
|
1640
|
+
catch {
|
|
1641
|
+
// A budget measurement must never break the turn.
|
|
1642
|
+
}
|
|
1643
|
+
// The instant this turn's model walk began. A failed generation records the
|
|
1644
|
+
// WALK (below) relative to this mark, so the trace can say what was tried.
|
|
1645
|
+
const modelWalkMark = Date.now();
|
|
1362
1646
|
const seenStepDigests = new Set();
|
|
1363
1647
|
const digestPrompt = (p) => {
|
|
1364
1648
|
try {
|
|
@@ -1490,6 +1774,36 @@ export class ChatCommand extends BaseCommand {
|
|
|
1490
1774
|
// (`answerQualityError`), which the loop rethrows once every candidate has
|
|
1491
1775
|
// narrated.
|
|
1492
1776
|
logger.error(String(err));
|
|
1777
|
+
// DIAGNOSABILITY — a failed turn used to record only the LAST attempt's
|
|
1778
|
+
// provider/model beside the FIRST error, so a reader could not tell which
|
|
1779
|
+
// model actually ran, nor WHY other models were not used. Record the WALK:
|
|
1780
|
+
// what was tried, what was parked (and for how long), and how large the
|
|
1781
|
+
// eligible pool was — the facts that separate a real shortage from a
|
|
1782
|
+
// routing gap. This is the exact ambiguity in the traces that prompted it:
|
|
1783
|
+
// a step labelled `local/qwen2.5:0.5b` carrying Gemini's 429.
|
|
1784
|
+
try {
|
|
1785
|
+
const report = modelBreadthReport(modelWalkMark, this.configManager);
|
|
1786
|
+
const tried = report.tried.filter((a) => !a.skipped);
|
|
1787
|
+
const parked = report.parked.filter((r) => r.active);
|
|
1788
|
+
const triedList = tried
|
|
1789
|
+
.slice(0, 6)
|
|
1790
|
+
.map((a) => `${a.provider}/${a.model} (${a.reason})`)
|
|
1791
|
+
.join(', ');
|
|
1792
|
+
const parkedList = parked
|
|
1793
|
+
.slice(0, 6)
|
|
1794
|
+
.map((r) => `${r.provider}${r.model ? `/${r.model}` : ''} (${r.kind})`)
|
|
1795
|
+
.join(', ');
|
|
1796
|
+
recordTraceEvent(chatTraceId, {
|
|
1797
|
+
kind: 'failover',
|
|
1798
|
+
summary: `the model layer failed — eligible pool ${report.poolSize ?? '?'} model(s) across ` +
|
|
1799
|
+
`${report.poolProviders ?? '?'} provider(s), ${tried.length} tried, ${parked.length} parked` +
|
|
1800
|
+
(triedList ? `; tried: ${triedList}` : '') +
|
|
1801
|
+
(parkedList ? `; parked: ${parkedList}` : ''),
|
|
1802
|
+
});
|
|
1803
|
+
}
|
|
1804
|
+
catch {
|
|
1805
|
+
// Diagnosis is a courtesy — it must never break the failure path.
|
|
1806
|
+
}
|
|
1493
1807
|
endTrace(chatTraceId, false, { kind: 'failed' });
|
|
1494
1808
|
result = {
|
|
1495
1809
|
// Sanitized on purpose: this content is delivered verbatim by every
|
|
@@ -1601,11 +1915,29 @@ export class ChatCommand extends BaseCommand {
|
|
|
1601
1915
|
if (result.content.trim() && !result.generationFailed && !result.cancelled) {
|
|
1602
1916
|
if (cacheEnabled) {
|
|
1603
1917
|
try {
|
|
1604
|
-
//
|
|
1605
|
-
//
|
|
1606
|
-
//
|
|
1607
|
-
//
|
|
1608
|
-
|
|
1918
|
+
// #30 — NEVER cache a turn that carries an honesty flag. The cache
|
|
1919
|
+
// stores text only, so storing a flagged answer would let a later
|
|
1920
|
+
// identical prompt (or the same prompt on another surface) replay a
|
|
1921
|
+
// known-unverified claim as a clean one — a truthfulness hole the whole
|
|
1922
|
+
// flag system exists to close. A flagged turn re-derives instead.
|
|
1923
|
+
if (turnCarriesHonestyFlag(result)) {
|
|
1924
|
+
debugLog?.event('cache.skip', { reason: 'honesty-flag', scope: turnScope });
|
|
1925
|
+
}
|
|
1926
|
+
else {
|
|
1927
|
+
// Keyed by the model that ACTUALLY answered (tryGenerate records it
|
|
1928
|
+
// on success), so a weak model's reply is never replayed as a strong
|
|
1929
|
+
// model's. `cacheModel` is the pre-flight fallback for the paths that
|
|
1930
|
+
// never resolve one (e.g. a cached-hit turn).
|
|
1931
|
+
//
|
|
1932
|
+
// #30 — the turn ACTIVITY rides with the text so a replay reports what
|
|
1933
|
+
// the cached turn did (tool cards, bounded), rather than reading as a
|
|
1934
|
+
// turn that did nothing.
|
|
1935
|
+
await cache.set(message, result.content, this.cacheModelFor(session) || cacheModel, session.type, undefined, turnScope, {
|
|
1936
|
+
...(result.toolCalls ? { toolCalls: result.toolCalls } : {}),
|
|
1937
|
+
...(result.successfulToolCalls ? { successfulToolCalls: result.successfulToolCalls } : {}),
|
|
1938
|
+
...(result.bounded ? { bounded: true } : {}),
|
|
1939
|
+
});
|
|
1940
|
+
}
|
|
1609
1941
|
}
|
|
1610
1942
|
catch {
|
|
1611
1943
|
// Best-effort.
|
|
@@ -1702,6 +2034,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1702
2034
|
transport: result.transport,
|
|
1703
2035
|
// WS1 — the findings this turn recorded, with their verdicts.
|
|
1704
2036
|
findings,
|
|
2037
|
+
// E — the honest "what happened" facts the TurnReport is derived from.
|
|
2038
|
+
successfulToolCalls: result.successfulToolCalls,
|
|
2039
|
+
runTrace: result.runTrace,
|
|
2040
|
+
unverifiedEdit: result.unverifiedEdit,
|
|
2041
|
+
unverifiedEditClaim: result.unverifiedEditClaim,
|
|
2042
|
+
noActionTaken: result.noActionTaken,
|
|
2043
|
+
// E-trace — the trace this turn was recorded under, so `answerOnce` can
|
|
2044
|
+
// attach the TurnReport to it (see `recordTurnReport`).
|
|
2045
|
+
traceId: chatTraceId,
|
|
1705
2046
|
});
|
|
1706
2047
|
}
|
|
1707
2048
|
/**
|
|
@@ -1735,7 +2076,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1735
2076
|
return `${session.type}:unresolved`;
|
|
1736
2077
|
}
|
|
1737
2078
|
}
|
|
1738
|
-
buildToolCallModel(message, session, options, mode, onToken, signal
|
|
2079
|
+
buildToolCallModel(message, session, options, mode, onToken, signal,
|
|
2080
|
+
/**
|
|
2081
|
+
* Intent-aware failover: the prior software ask when this turn is a bare
|
|
2082
|
+
* continuation ("yes", "do it"), computed by the caller from the history
|
|
2083
|
+
* it owns. The initial route used it; the mid-turn failover walk needs it
|
|
2084
|
+
* too, or a continuation whose first cloud provider dies re-routes on the
|
|
2085
|
+
* signal-free message and can land on a tiny local model.
|
|
2086
|
+
*/
|
|
2087
|
+
routingText) {
|
|
1739
2088
|
return async (messages, schemas, stepOnToken, stepSignal) => {
|
|
1740
2089
|
// The effective token sink: the caller's stream wins; when a step-level
|
|
1741
2090
|
// sink is also given (loop passthrough) they are the same channel.
|
|
@@ -1862,11 +2211,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
1862
2211
|
// content delivered as a single chunk so the typewriter channel
|
|
1863
2212
|
// still receives the answer (appears at once — today's behavior).
|
|
1864
2213
|
if (sink && typeof prov.generateToolsStream === 'function') {
|
|
1865
|
-
const result = await prov.generateToolsStream(messages, schemas, { ...options, model: effectiveModel, signal: abort }, sink);
|
|
2214
|
+
const result = await prov.generateToolsStream(messages, schemas, { ...options, model: effectiveModel, signal: abort, reasoningEffort: capabilityReasoningEffort(this.configManager) }, sink);
|
|
1866
2215
|
confuseCheck(result.content, result.toolCalls.length > 0);
|
|
1867
2216
|
return answered({ ...result, transport: 'native' });
|
|
1868
2217
|
}
|
|
1869
|
-
const result = await prov.generateTools(messages, schemas, { ...options, model: effectiveModel, signal: abort });
|
|
2218
|
+
const result = await prov.generateTools(messages, schemas, { ...options, model: effectiveModel, signal: abort, reasoningEffort: capabilityReasoningEffort(this.configManager) });
|
|
1870
2219
|
confuseCheck(result.content, result.toolCalls.length > 0);
|
|
1871
2220
|
if (sink && result.content)
|
|
1872
2221
|
sink(result.content);
|
|
@@ -1917,11 +2266,11 @@ export class ChatCommand extends BaseCommand {
|
|
|
1917
2266
|
let raw;
|
|
1918
2267
|
if (typeof prov.generateStream === 'function') {
|
|
1919
2268
|
const chunks = [];
|
|
1920
|
-
await prov.generateStream(prompt, { ...options, model: effectiveModel, signal: abort }, (t) => chunks.push(t));
|
|
2269
|
+
await prov.generateStream(prompt, { ...options, model: effectiveModel, signal: abort, reasoningEffort: capabilityReasoningEffort(this.configManager) }, (t) => chunks.push(t));
|
|
1921
2270
|
raw = chunks.join('');
|
|
1922
2271
|
}
|
|
1923
2272
|
else {
|
|
1924
|
-
raw = await prov.generate(prompt, { ...options, model: effectiveModel, signal: abort });
|
|
2273
|
+
raw = await prov.generate(prompt, { ...options, model: effectiveModel, signal: abort, reasoningEffort: capabilityReasoningEffort(this.configManager) });
|
|
1925
2274
|
}
|
|
1926
2275
|
const { text, calls } = extractFallbackToolCalls(raw);
|
|
1927
2276
|
confuseCheck(text, calls.length > 0);
|
|
@@ -1945,13 +2294,27 @@ export class ChatCommand extends BaseCommand {
|
|
|
1945
2294
|
// success log per landed candidate.
|
|
1946
2295
|
logger.warn(` ⚠️ ${session.provider.name} failed — trying the next auto candidate...`);
|
|
1947
2296
|
const failed = new Set([session.type]);
|
|
2297
|
+
// Intent-aware escalation for the FAILOVER walk (the live junk path):
|
|
2298
|
+
// a mid-turn failure used to re-route on the bare message alone, so a
|
|
2299
|
+
// software turn whose first cloud provider died fell through to a tiny
|
|
2300
|
+
// local model (`local/qwen2.5:0.5b`) that FABRICATED tool output
|
|
2301
|
+
// (trace-1791118650644-d73hyr). Pass the same prior-software-ask hint
|
|
2302
|
+
// the initial route used, so the router's agentic capability floor
|
|
2303
|
+
// applies here too — but NEVER drop a candidate the floor would keep,
|
|
2304
|
+
// so auto still cannot dead-end (local stays the last resort).
|
|
2305
|
+
const failoverRoutingText = routingText;
|
|
1948
2306
|
// Try ALL ranked candidates (no 3-candidate cap) — bounded by
|
|
1949
2307
|
// the number of known providers to prevent infinite loops.
|
|
1950
2308
|
const maxAttempts = 10;
|
|
1951
2309
|
for (let i = 0; i < maxAttempts; i++) {
|
|
1952
2310
|
let next = null;
|
|
1953
2311
|
try {
|
|
1954
|
-
next =
|
|
2312
|
+
next = failoverRoutingText
|
|
2313
|
+
? await this.routeMessageAuto(message, [...failed], {
|
|
2314
|
+
routingText: failoverRoutingText,
|
|
2315
|
+
fallbackFrom: firstType,
|
|
2316
|
+
})
|
|
2317
|
+
: await this.routeMessageAuto(message, [...failed], { fallbackFrom: firstType });
|
|
1955
2318
|
}
|
|
1956
2319
|
catch {
|
|
1957
2320
|
break;
|
|
@@ -1959,6 +2322,17 @@ export class ChatCommand extends BaseCommand {
|
|
|
1959
2322
|
if (!next || next.type === session.type || failed.has(next.type))
|
|
1960
2323
|
break;
|
|
1961
2324
|
failed.add(next.type);
|
|
2325
|
+
// Last-resort truthfulness: if even the agentic floor could not
|
|
2326
|
+
// avoid a weak model, say so ONCE so a degraded answer is never
|
|
2327
|
+
// mistaken for a real one (the tiny local model fabricated tool
|
|
2328
|
+
// results instead of admitting it could not run them). C2 — the
|
|
2329
|
+
// wording comes from the SHARED helper so chat, the orchestrator
|
|
2330
|
+
// and the dashboard cannot describe this three different ways.
|
|
2331
|
+
if (!isAgenticCapableModel(next.model, next.type)) {
|
|
2332
|
+
const notice = weakRouteNotice({ provider: next.type, model: next.model }, true) ??
|
|
2333
|
+
`⚠️ no agentic-capable model left — falling back to ${next.type}/${next.model}`;
|
|
2334
|
+
logger.warn(` no agentic-capable model left — ${notice}`);
|
|
2335
|
+
}
|
|
1962
2336
|
// Opt-in confirmation (routing.promptOnFailover): 'manual' stops
|
|
1963
2337
|
// the walk and lets the caller's error recovery handle it. Gated on
|
|
1964
2338
|
// an interactive stdin (inherited from the shared single-shot
|
|
@@ -2210,6 +2584,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
2210
2584
|
* so Auto routing never sends a request to a provider that would 401.
|
|
2211
2585
|
*/
|
|
2212
2586
|
async routeMessageAuto(message, excludeProviders = [], opts) {
|
|
2587
|
+
// The text ROUTING is decided from — the prior software ask for a bare
|
|
2588
|
+
// continuation, otherwise the message itself. `message` is used everywhere
|
|
2589
|
+
// else (the answer, the tool loop).
|
|
2590
|
+
const taskText = opts?.routingText?.trim() ? opts.routingText : message;
|
|
2213
2591
|
// Feed the SHARED circuit breaker into the router so a provider that has
|
|
2214
2592
|
// failed repeatedly (recorded by recordFailure below) is deprioritized by
|
|
2215
2593
|
// scoring, not just skipped by the candidate walk.
|
|
@@ -2227,7 +2605,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2227
2605
|
// circuit-breaker state on top.
|
|
2228
2606
|
// C3: the NLU parser seeds the router task-intent (same vocabulary every
|
|
2229
2607
|
// action command derives from resolveDispatch) when confident.
|
|
2230
|
-
const parsed = parseRequestSync(
|
|
2608
|
+
const parsed = parseRequestSync(taskText);
|
|
2231
2609
|
const dispatch = resolveDispatch(parsed);
|
|
2232
2610
|
// Routing decision cache (assessment v4 Phase 2): the loop engine
|
|
2233
2611
|
// resolves per turn; turns with identical STABLE routing inputs (intent,
|
|
@@ -2251,7 +2629,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2251
2629
|
const cacheSignature = routingCacheSignature([
|
|
2252
2630
|
'chat',
|
|
2253
2631
|
dispatch.taskIntentHint ?? null,
|
|
2254
|
-
analyzeComplexity(
|
|
2632
|
+
analyzeComplexity(taskText),
|
|
2255
2633
|
routingCfg.preferenceMode ?? null,
|
|
2256
2634
|
routingCfg.bandit === false ? 'b' : 'B',
|
|
2257
2635
|
routingCfg.mlRouter === true ? 'm' : 'M',
|
|
@@ -2264,7 +2642,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2264
2642
|
.sort()
|
|
2265
2643
|
.join(','),
|
|
2266
2644
|
]);
|
|
2267
|
-
const decision = withRoutingCache(cacheSignature, 30_000, () => getAutoRouter().resolve('chat',
|
|
2645
|
+
const decision = withRoutingCache(cacheSignature, 30_000, () => getAutoRouter().resolve('chat', taskText, {
|
|
2268
2646
|
...buildAutoResolveOptions(this.configManager, {
|
|
2269
2647
|
verbose: envBuff('DEBUG') === 'true',
|
|
2270
2648
|
contextHintTokens: opts?.contextHintTokens,
|
|
@@ -2280,6 +2658,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
2280
2658
|
score: decision.score,
|
|
2281
2659
|
complexity: String(decision.complexity),
|
|
2282
2660
|
explanation: decision.explanation,
|
|
2661
|
+
// A1/A2/C3 — carry the capability verdict + override reason so the trace
|
|
2662
|
+
// and console can show WHY the pair was chosen, not just which pair.
|
|
2663
|
+
agenticCapable: decision.agenticCapable,
|
|
2664
|
+
overrideReason: decision.overrideReason,
|
|
2283
2665
|
};
|
|
2284
2666
|
// Walk the ranked candidates (winner first) and return the first available
|
|
2285
2667
|
// provider — never a provider that lacks a key or endpoint. Providers that
|
|
@@ -2427,6 +2809,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2427
2809
|
source: 'chat',
|
|
2428
2810
|
agentType: 'chat',
|
|
2429
2811
|
task: message,
|
|
2812
|
+
verifyOnDemand: isMaxCapability(this.configManager),
|
|
2430
2813
|
})).model;
|
|
2431
2814
|
// Record the actually-used route for the dashboard audit trail
|
|
2432
2815
|
recordRoutingDecision({
|
|
@@ -2437,6 +2820,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
2437
2820
|
provider: candidate.provider,
|
|
2438
2821
|
model,
|
|
2439
2822
|
score: decision.score,
|
|
2823
|
+
agenticCapable: isAgenticCapableModel(model, candidate.provider),
|
|
2824
|
+
overrideReason: decision.overrideReason,
|
|
2825
|
+
...(opts?.fallbackFrom ? { fallbackFrom: opts.fallbackFrom } : {}),
|
|
2440
2826
|
});
|
|
2441
2827
|
return {
|
|
2442
2828
|
type: resolved.type,
|
|
@@ -2445,6 +2831,12 @@ export class ChatCommand extends BaseCommand {
|
|
|
2445
2831
|
ranked: candidates,
|
|
2446
2832
|
complexity: decision.complexity,
|
|
2447
2833
|
score: decision.score,
|
|
2834
|
+
agenticCapable: isAgenticCapableModel(model, resolved.type),
|
|
2835
|
+
overrideReason: decision.overrideReason,
|
|
2836
|
+
taskProfile: {
|
|
2837
|
+
intent: decision.taskProfile.intent,
|
|
2838
|
+
requiresVerification: decision.taskProfile.requiresVerification,
|
|
2839
|
+
},
|
|
2448
2840
|
};
|
|
2449
2841
|
}
|
|
2450
2842
|
}
|
|
@@ -2468,6 +2860,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
2468
2860
|
provider: usableProvider,
|
|
2469
2861
|
model: decision.model,
|
|
2470
2862
|
score: decision.score,
|
|
2863
|
+
agenticCapable: isAgenticCapableModel(decision.model, usableProvider),
|
|
2864
|
+
overrideReason: decision.overrideReason,
|
|
2865
|
+
...(opts?.fallbackFrom ? { fallbackFrom: opts.fallbackFrom } : {}),
|
|
2471
2866
|
});
|
|
2472
2867
|
const resolved = resolveProvider(this.configManager, usableProvider);
|
|
2473
2868
|
const model = (await resolveRoute({
|
|
@@ -2477,6 +2872,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
2477
2872
|
source: 'chat',
|
|
2478
2873
|
agentType: 'chat',
|
|
2479
2874
|
task: message,
|
|
2875
|
+
verifyOnDemand: isMaxCapability(this.configManager),
|
|
2480
2876
|
})).model;
|
|
2481
2877
|
return {
|
|
2482
2878
|
type: resolved.type,
|