agent-nuvira 3.1.3 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/agents/reasoner.d.ts +8 -0
- package/dist/agents/agents/reasoner.d.ts.map +1 -1
- package/dist/agents/agents/reasoner.js +53 -2
- package/dist/agents/agents/reasoner.js.map +1 -1
- package/dist/agents/agents/writer.d.ts +24 -0
- package/dist/agents/agents/writer.d.ts.map +1 -1
- package/dist/agents/agents/writer.js +200 -0
- package/dist/agents/agents/writer.js.map +1 -1
- package/dist/agents/composite-plan.d.ts +143 -0
- package/dist/agents/composite-plan.d.ts.map +1 -0
- package/dist/agents/composite-plan.js +399 -0
- package/dist/agents/composite-plan.js.map +1 -0
- package/dist/agents/long-form-plan.d.ts +156 -0
- package/dist/agents/long-form-plan.d.ts.map +1 -0
- package/dist/agents/long-form-plan.js +274 -0
- package/dist/agents/long-form-plan.js.map +1 -0
- package/dist/agents/orchestrator.d.ts +107 -0
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +501 -35
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/agents/prompt-assembly.d.ts +8 -0
- package/dist/agents/prompt-assembly.d.ts.map +1 -1
- package/dist/agents/prompt-assembly.js +17 -0
- package/dist/agents/prompt-assembly.js.map +1 -1
- package/dist/cli/chat.d.ts +33 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +161 -7
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/execute.d.ts +12 -0
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +150 -2
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/loop-executor.d.ts +17 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +160 -1
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/models.d.ts.map +1 -1
- package/dist/cli/models.js +10 -0
- package/dist/cli/models.js.map +1 -1
- package/dist/cli/trace.d.ts +19 -0
- package/dist/cli/trace.d.ts.map +1 -1
- package/dist/cli/trace.js +109 -2
- package/dist/cli/trace.js.map +1 -1
- package/dist/gateway/registry.d.ts +31 -0
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +162 -13
- package/dist/gateway/registry.js.map +1 -1
- package/dist/inference/model-entitlement.d.ts +46 -0
- package/dist/inference/model-entitlement.d.ts.map +1 -0
- package/dist/inference/model-entitlement.js +98 -0
- package/dist/inference/model-entitlement.js.map +1 -0
- package/dist/inference/tool-call-utils.d.ts +10 -0
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +90 -0
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/learning/autonomy-policy.d.ts +370 -0
- package/dist/learning/autonomy-policy.d.ts.map +1 -0
- package/dist/learning/autonomy-policy.js +544 -0
- package/dist/learning/autonomy-policy.js.map +1 -0
- package/dist/learning/cost-tracker.d.ts +14 -0
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +22 -0
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/credential-fingerprint.d.ts +58 -0
- package/dist/learning/credential-fingerprint.d.ts.map +1 -0
- package/dist/learning/credential-fingerprint.js +126 -0
- package/dist/learning/credential-fingerprint.js.map +1 -0
- package/dist/learning/deliverable-class.d.ts +178 -0
- package/dist/learning/deliverable-class.d.ts.map +1 -0
- package/dist/learning/deliverable-class.js +504 -0
- package/dist/learning/deliverable-class.js.map +1 -0
- package/dist/learning/engine-router.d.ts +12 -1
- package/dist/learning/engine-router.d.ts.map +1 -1
- package/dist/learning/engine-router.js +28 -0
- package/dist/learning/engine-router.js.map +1 -1
- package/dist/learning/long-form.d.ts +252 -0
- package/dist/learning/long-form.d.ts.map +1 -0
- package/dist/learning/long-form.js +521 -0
- package/dist/learning/long-form.js.map +1 -0
- package/dist/learning/model-first-router.d.ts +17 -0
- package/dist/learning/model-first-router.d.ts.map +1 -1
- package/dist/learning/model-first-router.js +27 -0
- package/dist/learning/model-first-router.js.map +1 -1
- package/dist/learning/model-registry.d.ts +66 -4
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +67 -6
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/model-warmup.d.ts +97 -2
- package/dist/learning/model-warmup.d.ts.map +1 -1
- package/dist/learning/model-warmup.js +165 -60
- package/dist/learning/model-warmup.js.map +1 -1
- package/dist/learning/prompt-layers.d.ts +61 -0
- package/dist/learning/prompt-layers.d.ts.map +1 -0
- package/dist/learning/prompt-layers.js +140 -0
- package/dist/learning/prompt-layers.js.map +1 -0
- package/dist/learning/provider-limits.d.ts +66 -0
- package/dist/learning/provider-limits.d.ts.map +1 -0
- package/dist/learning/provider-limits.js +184 -0
- package/dist/learning/provider-limits.js.map +1 -0
- package/dist/learning/reasoning-trace.d.ts +125 -1
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +85 -2
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts +36 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +80 -5
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/unattended-job.d.ts +350 -0
- package/dist/learning/unattended-job.d.ts.map +1 -0
- package/dist/learning/unattended-job.js +636 -0
- package/dist/learning/unattended-job.js.map +1 -0
- package/dist/learning/unattended-progress.d.ts +95 -0
- package/dist/learning/unattended-progress.d.ts.map +1 -0
- package/dist/learning/unattended-progress.js +147 -0
- package/dist/learning/unattended-progress.js.map +1 -0
- package/dist/learning/working-state.d.ts +109 -0
- package/dist/learning/working-state.d.ts.map +1 -0
- package/dist/learning/working-state.js +244 -0
- package/dist/learning/working-state.js.map +1 -0
- package/dist/nlu/conversation-gate.d.ts +33 -0
- package/dist/nlu/conversation-gate.d.ts.map +1 -1
- package/dist/nlu/conversation-gate.js +68 -3
- package/dist/nlu/conversation-gate.js.map +1 -1
- package/dist/tools/coding-tools.d.ts.map +1 -1
- package/dist/tools/coding-tools.js +95 -19
- package/dist/tools/coding-tools.js.map +1 -1
- package/dist/tools/edit-verification.d.ts +142 -0
- package/dist/tools/edit-verification.d.ts.map +1 -0
- package/dist/tools/edit-verification.js +262 -0
- package/dist/tools/edit-verification.js.map +1 -0
- package/dist/tools/git-tool.d.ts.map +1 -1
- package/dist/tools/git-tool.js +37 -4
- package/dist/tools/git-tool.js.map +1 -1
- package/dist/tools/registry.d.ts +30 -2
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +40 -11
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/run-cli.d.ts.map +1 -1
- package/dist/tools/run-cli.js +40 -13
- package/dist/tools/run-cli.js.map +1 -1
- package/dist/tools/run-terminal.d.ts +6 -1
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +158 -18
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +109 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +362 -2
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +8 -0
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +9 -0
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/chat-retry.d.ts.map +1 -1
- package/dist/web-dashboard/chat-retry.js +10 -2
- package/dist/web-dashboard/chat-retry.js.map +1 -1
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +80 -8
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +81 -4
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/package.json +1 -1
- package/src/web-dashboard/public/assets/{index-Co7Hk2FT.js → index-kCUkORm7.js} +2 -2
- package/src/web-dashboard/public/assets/{index-Co7Hk2FT.js.map → index-kCUkORm7.js.map} +1 -1
- package/src/web-dashboard/public/index.html +1 -1
package/dist/cli/chat.d.ts
CHANGED
|
@@ -147,6 +147,22 @@ export declare class ChatCommand extends BaseCommand {
|
|
|
147
147
|
* failing into dead ends — the fire-and-forget keeps the first message fast.
|
|
148
148
|
*/
|
|
149
149
|
private coldStartProbeFired;
|
|
150
|
+
/**
|
|
151
|
+
* G5 — TRACE FIDELITY. The Auto-router decision snapshot for the current
|
|
152
|
+
* turn, stashed at resolve time so the reasoning trace can record WHY a step
|
|
153
|
+
* used the provider/model it did. The audit of the calculator session found
|
|
154
|
+
* 98/99 steps stamped with the session default `gemini/gemini-3.1-flash-lite`
|
|
155
|
+
* and **0/99** steps carrying a routing snapshot — so the Trace tab's "model
|
|
156
|
+
* used" column was really "the session's default model".
|
|
157
|
+
*/
|
|
158
|
+
private lastRouteSnapshot?;
|
|
159
|
+
/**
|
|
160
|
+
* G5 — the provider/model of the LAST generation attempt (post-failover,
|
|
161
|
+
* post-default-resolution). Read by the chat trace recorder so a step names
|
|
162
|
+
* the model that actually ran it, instead of whatever the session default
|
|
163
|
+
* happened to be.
|
|
164
|
+
*/
|
|
165
|
+
private lastAttempt?;
|
|
150
166
|
/**
|
|
151
167
|
* P3 — programmatic single-turn answer for the dashboard chat console.
|
|
152
168
|
*
|
|
@@ -209,6 +225,17 @@ export declare class ChatCommand extends BaseCommand {
|
|
|
209
225
|
* project, cwd-aware).
|
|
210
226
|
*/
|
|
211
227
|
projectContext?: string;
|
|
228
|
+
/**
|
|
229
|
+
* Session 3 — CHANNEL / FORMAT POLICY for the STABLE layer.
|
|
230
|
+
*
|
|
231
|
+
* The gateway used to prepend the "RESPONSE FORMAT (non-negotiable…)" block
|
|
232
|
+
* to every INBOUND USER TURN — i.e. the same policy was re-injected as
|
|
233
|
+
* volatile content on every message (~80% of the visible user turn on
|
|
234
|
+
* WhatsApp turns). Policy is identical every turn, so it belongs in the
|
|
235
|
+
* system prompt: it is then byte-stable (prompt-cacheable) and shows up in
|
|
236
|
+
* the layered trace's `systemDigest` instead of polluting the ask.
|
|
237
|
+
*/
|
|
238
|
+
systemPolicy?: string;
|
|
212
239
|
/**
|
|
213
240
|
* P4 — the attached project's directory. When set, the turn ALSO recalls
|
|
214
241
|
* that project's prior sessions + facts (`autoRecall`) and injects them
|
|
@@ -252,6 +279,12 @@ export declare class ChatCommand extends BaseCommand {
|
|
|
252
279
|
* carried out. Surfaces must not present such a turn as "in progress".
|
|
253
280
|
*/
|
|
254
281
|
unfulfilledPromise?: boolean;
|
|
282
|
+
/**
|
|
283
|
+
* G13b — the request asked for an authored deliverable to be produced and
|
|
284
|
+
* the turn wrote nothing to disk. The reply may be excellent prose; the
|
|
285
|
+
* artifact does not exist, so no surface may read it as finished work.
|
|
286
|
+
*/
|
|
287
|
+
undeliveredArtifact?: boolean;
|
|
255
288
|
provider?: string;
|
|
256
289
|
model?: string;
|
|
257
290
|
}>;
|
package/dist/cli/chat.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"chat.d.ts","sourceRoot":"","sources":["../../src/cli/chat.ts"],"names":[],"mappings":"AAIA,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AAEpC,OAAO,EAAE,WAAW,EAAc,MAAM,eAAe,CAAC;
|
|
1
|
+
{"version":3,"file":"chat.d.ts","sourceRoot":"","sources":["../../src/cli/chat.ts"],"names":[],"mappings":"AAIA,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AAEpC,OAAO,EAAE,WAAW,EAAc,MAAM,eAAe,CAAC;AAiCxD,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,kBAAkB,CAAC;AAqBtD;;;;;GAKG;AACH,MAAM,WAAW,YAAY;IAC3B,EAAE,CAAC,EAAE,MAAM,CAAC;IACZ,IAAI,EAAE,MAAM,CAAC;IACb,IAAI,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAC/B,EAAE,CAAC,EAAE,OAAO,CAAC;IACb,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,UAAU,CAAC,EAAE,MAAM,CAAC;CACrB;AAED,OAAO,EAA+B,KAAK,WAAW,EAAE,MAAM,sBAAsB,CAAC;AACrF,OAAO,EAGL,KAAK,kBAAkB,EACxB,MAAM,4BAA4B,CAAC;AA0JpC,sEAAsE;AACtE,MAAM,WAAW,wBAAwB;IACvC,oDAAoD;IACpD,QAAQ,EAAE,OAAO,CAAC;IAClB,0EAA0E;IAC1E,WAAW,EAAE,OAAO,CAAC;CACtB;AAED,oEAAoE;AACpE,MAAM,WAAW,yBAAyB;IACxC,GAAG,CAAC,EAAE,OAAO,CAAC;IACd,0EAA0E;IAC1E,IAAI,CAAC,EAAE,MAAM,CAAC;CACf;AAED;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,wBAAgB,uBAAuB,CACrC,MAAM,EAAE,aAAa,EACrB,IAAI,CAAC,EAAE,yBAAyB,GAC/B,wBAAwB,CA4B1B;AAID;;;;;;;;GAQG;AACH,wBAAsB,gBAAgB,CACpC,IAAI,EAAE,MAAM,EACZ,aAAa,EAAE,GAAG,EAClB,OAAO,CAAC,EAAE;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;IAAC,KAAK,CAAC,EAAE,MAAM,CAAA;CAAE,GAC9C,OAAO,CAAC,IAAI,CAAC,CAYf;AAED;;;;;;;;;GASG;AACH;;;;;GAKG;AACH,eAAO,MAAM,yBAAyB,uBAA0B,CAAC;AAEjE;;;;;;;;;;;;;GAaG;AACH,wBAAsB,0BAA0B,CAAC,CAAC,EAChD,OAAO,EAAE,MAAM,OAAO,CAAC,CAAC,CAAC,EACzB,MAAM,CAAC,EAAE,WAAW,EACpB,OAAO,CAAC,EAAE,CAAC,aAAa,EAAE,MAAM,EAAE,GAAG,EAAE,OAAO,KAAK,IAAI,GACtD,OAAO,CAAC,CAAC,CAAC,CAWZ;AAkCD,qBAAa,WAAY,SAAQ,WAAW;IAC1C,OAAO,CAAC,WAAW,CAAS;IAE5B;;;;;;;;;;;;;OAaG;IACH,OAAO,CAAC,sBAAsB,CAA6B;IAE3D;;;;;;;;;OASG;IACH,OAAO,CAAC,mBAAmB,CAA6B;IAMxD;;;;;;OAMG;IACH,OAAO,CAAC,+BAA+B,CAAqB;IAE5D;;;;OAIG;IACH,OAAO,CAAC,SAAS,CAAmE;IAEpF;;;;;OAKG;IACH,OAAO,CAAC,mBAAmB,CAAS;IAEpC;;;;;;;OAOG;IACH,OAAO,CAAC,iBAAiB,CAAC,CAAgE;IAE1F;;;;;OAKG;IACH,OAAO,CAAC,WAAW,CAAC,CAAuC;IAE3D;;;;;;;;;OASG;IACG,UAAU,CACd,OAAO,EAAE,MAAM,EACf,IAAI,GAAE;QACJ,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,KAAK,CAAC,EAAE,MAAM,CAAC;QACf,GAAG,CAAC,EAAE,OAAO,CAAC;QAAM,OAAO,CAAC,EAAE,KAAK,CAAC;YAAE,IAAI,EAAE,MAAM,CAAC;YAAC,OAAO,EAAE,MAAM,CAAA;SAAE,CAAC,CAAC;QACzE,OAAO,CAAC,EAAE,WAAW,CAAC,SAAS,CAAC,CAAC;QACjC;;;;;WAKG;QACH,YAAY,CAAC,EAAE,OAAO,CAAC;QACvB,+DAA+D;QAC/D,UAAU,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,IAAI,CAAC;QACpC;;;;WAIG;QACH,UAAU,CAAC,EAAE,CAAC,KAAK,EAAE,SAAS,GAAG,QAAQ,EAAE,IAAI,EAAE,YAAY,KAAK,IAAI,CAAC;QACvE;;;;WAIG,CAAI,YAAY,CAAC,EAAE,CAAC,QAAQ,EAAE,OAAO,wBAAwB,EAAE,YAAY,KAAK,IAAI,CAAC;QACxF;;;;WAIG;QACH,SAAS,CAAC,EAAE,CAAC,OAAO,EAAE,OAAO,sBAAsB,EAAE,cAAc,KAAK,IAAI,CAAC;QAC7E,4EAA4E;QAC5E,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,OAAO,wBAAwB,EAAE,iBAAiB,KAAK,IAAI,CAAC;QACrF;;;WAGG;QACH,SAAS,CAAC,EAAE,OAAO,wBAAwB,EAAE,aAAa,CAAC;QAC3D,iGAAiG;QACjG,OAAO,CAAC,EAAE,WAAW,CAAC,SAAS,CAAC,CAAC;QACjC;;;;;;WAMG;QACH,cAAc,CAAC,EAAE,MAAM,CAAC;QACxB;;;;;;;;;WASG;QACH,YAAY,CAAC,EAAE,MAAM,CAAC;QACtB;;;;;;;WAOG;QACH,WAAW,CAAC,EAAE,MAAM,CAAC;QACrB;;;;;WAKG;QACH,OAAO,CAAC,EAAE,CAAC,KAAK,EAAE,MAAM,KAAK,IAAI,CAAC;QAClC;;;;;WAKG;QACH,MAAM,CAAC,EAAE,WAAW,CAAC;KACjB,GACL,OAAO,CAAC;QACT,OAAO,EAAE,MAAM,CAAC;QAChB,SAAS,EAAE,kBAAkB,EAAE,CAAC;QAChC,gBAAgB,CAAC,EAAE,OAAO,CAAC;QAC3B,yEAAyE;QACzE,SAAS,CAAC,EAAE,OAAO,CAAC;QACpB,0EAA0E;QAC1E,OAAO,CAAC,EAAE,OAAO,CAAC;QAClB,4EAA4E;QAC5E,SAAS,CAAC,EAAE,MAAM,EAAE,CAAC;QACrB;;;WAGG;QACH,qBAAqB,CAAC,EAAE,OAAO,CAAC;QAChC;;;WAGG;QACH,kBAAkB,CAAC,EAAE,OAAO,CAAC;QAC7B;;;;WAIG;QACH,mBAAmB,CAAC,EAAE,OAAO,CAAC;QAC9B,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,KAAK,CAAC,EAAE,MAAM,CAAC;KAChB,CAAC;IAmHA,MAAM,IAAI,OAAO;YAgBH,OAAO;IAkXrB;;;;;;;;;;;;;OAaG;YACW,aAAa;IA0lB3B;;;;;;OAMG;IACH;;;;;;;;;;;;OAYG;IACH,OAAO,CAAC,aAAa;IAUrB,OAAO,CAAC,kBAAkB;IAsX1B;;;;OAIG;YACW,eAAe;IAkC7B;;;;;OAKG;IACH,OAAO,CAAC,cAAc;IAUtB;;;;;;;;;;;;;;;OAeG;IACH;;;;;;;;;;;OAWG;IACH,OAAO,CAAC,yBAAyB;YAgBnB,eAAe;IAI7B;;;;OAIG;IACH;;;;;;;OAOG;YACW,gBAAgB;IA4Q9B;;;;;;;;OAQG;IACH,OAAO,CAAC,kBAAkB;YAmFZ,aAAa;CAwG5B"}
|
package/dist/cli/chat.js
CHANGED
|
@@ -23,6 +23,7 @@ import { getAutoRouter, isAutoModel, isAutoProvider, governanceVerdict } from '.
|
|
|
23
23
|
import { estimateTokens } from '../learning/cost-tracker.js';
|
|
24
24
|
import { getModelRegistry } from '../learning/model-registry.js';
|
|
25
25
|
import { refreshModelRegistry } from '../inference/model-probe.js';
|
|
26
|
+
import { startWarmupDaemon } from '../learning/model-warmup.js';
|
|
26
27
|
import { recordRoutingDecision } from '../learning/routing-history.js';
|
|
27
28
|
import { shouldConfirmFailover, promptFailoverChoice } from './failover-prompt.js';
|
|
28
29
|
import { buildAutoResolveOptions } from '../learning/resolve-options.js';
|
|
@@ -35,7 +36,8 @@ import { resolveDispatch } from '../nlu/actions.js';
|
|
|
35
36
|
import { hasCodingAction, resolveAskKind } from '../nlu/conversation-gate.js';
|
|
36
37
|
import { runToolLoop, extractFallbackToolCalls } from '../tools/tool-loop.js';
|
|
37
38
|
import { detectAnswerQualityFailure, answerQualityError, toUserFacingGenerationError, isToolCallingUnsupported, stripToolCallArtifacts, } from '../inference/tool-call-utils.js';
|
|
38
|
-
import { beginTrace, endTrace, recordStep, buildTraceOutcome } from '../learning/reasoning-trace.js';
|
|
39
|
+
import { beginTrace, endTrace, recordStep, recordTraceEvent, buildTraceOutcome } from '../learning/reasoning-trace.js';
|
|
40
|
+
import { recordWorkingState, getWorkingState, formatWorkingState } from '../learning/working-state.js';
|
|
39
41
|
import { getLoopExposureMode } from '../tools/toolsets.js';
|
|
40
42
|
import { resolveModelHarnessProfile, shouldSkipNativeTools } from '../learning/model-harness.js';
|
|
41
43
|
import { resolveAdapterDefault, hasCredentials } from '../learning/model-selection.js';
|
|
@@ -347,6 +349,22 @@ export class ChatCommand extends BaseCommand {
|
|
|
347
349
|
* failing into dead ends — the fire-and-forget keeps the first message fast.
|
|
348
350
|
*/
|
|
349
351
|
coldStartProbeFired = false;
|
|
352
|
+
/**
|
|
353
|
+
* G5 — TRACE FIDELITY. The Auto-router decision snapshot for the current
|
|
354
|
+
* turn, stashed at resolve time so the reasoning trace can record WHY a step
|
|
355
|
+
* used the provider/model it did. The audit of the calculator session found
|
|
356
|
+
* 98/99 steps stamped with the session default `gemini/gemini-3.1-flash-lite`
|
|
357
|
+
* and **0/99** steps carrying a routing snapshot — so the Trace tab's "model
|
|
358
|
+
* used" column was really "the session's default model".
|
|
359
|
+
*/
|
|
360
|
+
lastRouteSnapshot;
|
|
361
|
+
/**
|
|
362
|
+
* G5 — the provider/model of the LAST generation attempt (post-failover,
|
|
363
|
+
* post-default-resolution). Read by the chat trace recorder so a step names
|
|
364
|
+
* the model that actually ran it, instead of whatever the session default
|
|
365
|
+
* happened to be.
|
|
366
|
+
*/
|
|
367
|
+
lastAttempt;
|
|
350
368
|
/**
|
|
351
369
|
* P3 — programmatic single-turn answer for the dashboard chat console.
|
|
352
370
|
*
|
|
@@ -429,7 +447,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
429
447
|
}
|
|
430
448
|
const parsed = parseRequestSync(message);
|
|
431
449
|
const dispatchDecision = resolvePipelineDispatch(parsed, { dev: opts.dev, text: message });
|
|
432
|
-
const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation });
|
|
450
|
+
const answer = await this.runChatAnswer(message, opts.history ?? [], { type, provider, model }, { provider: mergedOpts.provider, model: mergedOpts.model, dev: mergedOpts.dev, cache: true }, true, { auto: autoMode }, parsed, { askUser: opts.askUser, onProgress: opts.onProgress, onToolCall: opts.onToolCall, onPlanChange: opts.onPlanChange, onGitDiff: opts.onGitDiff, onSkillDraft: opts.onSkillDraft, planStore: opts.planStore ?? this.planStore, gateway: opts.gateway, projectContext: opts.projectContext, recallContext: recallBlock, projectPath: opts.projectPath, onToken: opts.onToken, signal: opts.signal, continuation: opts.continuation, systemPolicy: opts.systemPolicy });
|
|
433
451
|
// No-model fallback: the tool loop could not generate a single response
|
|
434
452
|
// AND the rules assessed a high-confidence pipeline intent — run the
|
|
435
453
|
// pipeline directly (rules decide only when the model is unavailable; the
|
|
@@ -456,6 +474,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
456
474
|
toolCalls: answer.toolCalls,
|
|
457
475
|
unverifiedActionClaim: answer.unverifiedActionClaim,
|
|
458
476
|
unfulfilledPromise: answer.unfulfilledPromise,
|
|
477
|
+
undeliveredArtifact: answer.undeliveredArtifact,
|
|
459
478
|
provider: type,
|
|
460
479
|
model,
|
|
461
480
|
};
|
|
@@ -884,8 +903,20 @@ export class ChatCommand extends BaseCommand {
|
|
|
884
903
|
ambientProjectContext = undefined;
|
|
885
904
|
}
|
|
886
905
|
}
|
|
906
|
+
// G4 — carry THIS project's working state (files changed, verification debt,
|
|
907
|
+
// user-reported regressions) into the turn, so the model does not re-derive
|
|
908
|
+
// what previous turns already established. This is the fix for the
|
|
909
|
+
// calculator session's core drift (it re-diagnosed the same root cause six
|
|
910
|
+
// times, then undid its own earlier fixes).
|
|
911
|
+
const workingStatePath = ctxOverrides?.projectPath || process.cwd();
|
|
912
|
+
const workingStateBlock = formatWorkingState(getWorkingState(workingStatePath));
|
|
913
|
+
// Session 3 — channel/format policy lives in the STABLE layer. It is
|
|
914
|
+
// identical on every message, so keeping it here makes the system prompt
|
|
915
|
+
// byte-stable (prompt-cacheable) AND removes it from the volatile user
|
|
916
|
+
// turn, where it used to occupy ~80% of the ask on WhatsApp turns.
|
|
917
|
+
const systemPolicyBlock = ctxOverrides?.systemPolicy ? `\n\n${ctxOverrides.systemPolicy}` : '';
|
|
887
918
|
const thread = [
|
|
888
|
-
{ role: 'system', content: systemText + skillHint },
|
|
919
|
+
{ role: 'system', content: systemText + systemPolicyBlock + skillHint },
|
|
889
920
|
// P3 — the attached project's bounded snapshot (path + file tree +
|
|
890
921
|
// symbol map) rides in before the conversation, exactly like --file
|
|
891
922
|
// context: the model knows what it is looking at without being told.
|
|
@@ -901,6 +932,8 @@ export class ChatCommand extends BaseCommand {
|
|
|
901
932
|
...(ctxOverrides?.recallContext
|
|
902
933
|
? [{ role: 'user', content: ctxOverrides.recallContext }]
|
|
903
934
|
: []),
|
|
935
|
+
// G4 — the deterministic working-state ledger (never a summary).
|
|
936
|
+
...(workingStateBlock ? [{ role: 'user', content: workingStateBlock }] : []),
|
|
904
937
|
...(fileContext
|
|
905
938
|
? [{ role: 'user', content: `[File context]\n${fileContext}` }]
|
|
906
939
|
: []),
|
|
@@ -925,6 +958,16 @@ export class ChatCommand extends BaseCommand {
|
|
|
925
958
|
// loader channel the tool writes and the loop reads. 'all' (default)
|
|
926
959
|
// keeps the pre-tiering behavior byte-identical.
|
|
927
960
|
const loadedExtraTools = new Set();
|
|
961
|
+
// G3 — the files this turn mutates, observed on the tool event stream so
|
|
962
|
+
// the ledger can remember them (the loop reports tool NAMES, not paths).
|
|
963
|
+
const touchedFiles = new Set();
|
|
964
|
+
/**
|
|
965
|
+
* G18 — the turn's trace id, for the autonomy-gate events that tools emit
|
|
966
|
+
* DURING the loop. Assigned a few lines below (the trace begins once the
|
|
967
|
+
* thread and tool surface exist); the emit can only fire from tool
|
|
968
|
+
* execution, which is strictly after that assignment.
|
|
969
|
+
*/
|
|
970
|
+
let traceIdForEvents;
|
|
928
971
|
const toolContext = {
|
|
929
972
|
configManager: this.configManager,
|
|
930
973
|
loadedExtraTools,
|
|
@@ -932,6 +975,16 @@ export class ChatCommand extends BaseCommand {
|
|
|
932
975
|
// agent operates inside the project (not the dashboard server's cwd).
|
|
933
976
|
cwd: ctxOverrides?.projectPath || process.cwd(),
|
|
934
977
|
emit: (event, data, source) => {
|
|
978
|
+
// G3 — collect mutated file paths from `tool:started` (which carries
|
|
979
|
+
// the arguments) so the working-state ledger knows what changed.
|
|
980
|
+
if (event === 'tool:started') {
|
|
981
|
+
const t = data;
|
|
982
|
+
if (t?.tool === 'edit_file' || t?.tool === 'write_file') {
|
|
983
|
+
const p = t.args?.path ?? t.args?.file_path ?? t.args?.file;
|
|
984
|
+
if (typeof p === 'string' && p)
|
|
985
|
+
touchedFiles.add(p);
|
|
986
|
+
}
|
|
987
|
+
}
|
|
935
988
|
// P0.6 — forward tool-call lifecycle events to the GUI before they
|
|
936
989
|
// reach the bus (the bus drives hooks; the override drives the card
|
|
937
990
|
// stream). Other events keep flowing to the bus untouched.
|
|
@@ -951,6 +1004,21 @@ export class ChatCommand extends BaseCommand {
|
|
|
951
1004
|
if (ctxOverrides?.onSkillDraft && event === 'skill:draft') {
|
|
952
1005
|
ctxOverrides.onSkillDraft(data);
|
|
953
1006
|
}
|
|
1007
|
+
// G18 — an autonomy gate DECIDING to proceed is a fact about the turn
|
|
1008
|
+
// ("this change was applied without asking, and here is why"), not just
|
|
1009
|
+
// a bus notification: record it on the turn's trace so the decision is
|
|
1010
|
+
// auditable after the fact.
|
|
1011
|
+
if (event === 'autonomy:write-applied' && traceIdForEvents) {
|
|
1012
|
+
const d = data;
|
|
1013
|
+
recordTraceEvent(traceIdForEvents, {
|
|
1014
|
+
kind: 'gate',
|
|
1015
|
+
gate: 'autonomy',
|
|
1016
|
+
...(d?.tool ? { tool: d.tool } : {}),
|
|
1017
|
+
summary: d?.reason
|
|
1018
|
+
? `proceeded without asking — ${d.reason}`
|
|
1019
|
+
: 'proceeded without asking — the request itself was the authorization',
|
|
1020
|
+
});
|
|
1021
|
+
}
|
|
954
1022
|
getEventBus().emit(event, data, source);
|
|
955
1023
|
},
|
|
956
1024
|
// P3 — the dashboard chat console injects a NON-TTY ask_user renderer
|
|
@@ -991,6 +1059,8 @@ export class ChatCommand extends BaseCommand {
|
|
|
991
1059
|
provider: session.type,
|
|
992
1060
|
model: session.model,
|
|
993
1061
|
});
|
|
1062
|
+
// G18 — the tool-context emit (declared above) now has somewhere to write.
|
|
1063
|
+
traceIdForEvents = chatTraceId;
|
|
994
1064
|
const seenStepDigests = new Set();
|
|
995
1065
|
const digestPrompt = (p) => {
|
|
996
1066
|
try {
|
|
@@ -1013,20 +1083,28 @@ export class ChatCommand extends BaseCommand {
|
|
|
1013
1083
|
const promptPreview = lastUserIdx !== -1
|
|
1014
1084
|
? `${prompt.slice(0, 80)}…\n\n${prompt.slice(lastUserIdx)}`.slice(0, 500)
|
|
1015
1085
|
: prompt.slice(0, 300);
|
|
1086
|
+
// G5 — the REAL per-call model/provider (post-failover) and the
|
|
1087
|
+
// Auto-router snapshot, plus estimateTokens (the old chars/4 estimate
|
|
1088
|
+
// misreported usage). Falls back to the session default only when no
|
|
1089
|
+
// attempt was recorded (e.g. a cached/offline step).
|
|
1016
1090
|
recordStep(chatTraceId, {
|
|
1017
1091
|
agentType: 'chat',
|
|
1018
1092
|
description: message.slice(0, 120),
|
|
1019
|
-
|
|
1020
|
-
|
|
1093
|
+
// Session 3 — the FULL prompt feeds the layered digests and the
|
|
1094
|
+
// one-time stable-layer capture (see `promptFull` in recordStep).
|
|
1095
|
+
promptFull: prompt,
|
|
1096
|
+
provider: this.lastAttempt?.provider || session.type,
|
|
1097
|
+
model: this.lastAttempt?.model || session.model || 'unknown',
|
|
1021
1098
|
promptDigest: digest,
|
|
1022
1099
|
promptPreview,
|
|
1023
1100
|
responsePreview: output.slice(0, 1000),
|
|
1024
1101
|
responseLength: output.length,
|
|
1025
|
-
inputTokens:
|
|
1026
|
-
outputTokens:
|
|
1102
|
+
inputTokens: estimateTokens(prompt),
|
|
1103
|
+
outputTokens: estimateTokens(output),
|
|
1027
1104
|
latencyMs,
|
|
1028
1105
|
success: ok,
|
|
1029
1106
|
error,
|
|
1107
|
+
...(mode.auto && this.lastRouteSnapshot ? { routing: this.lastRouteSnapshot } : {}),
|
|
1030
1108
|
});
|
|
1031
1109
|
}
|
|
1032
1110
|
catch {
|
|
@@ -1069,6 +1147,10 @@ export class ChatCommand extends BaseCommand {
|
|
|
1069
1147
|
maxParallelReads: harness.maxParallelReads,
|
|
1070
1148
|
onToken: ctxOverrides?.onToken,
|
|
1071
1149
|
signal: ctxOverrides?.signal,
|
|
1150
|
+
// G18 — the same sink the execute loop uses: tool calls, gate decisions
|
|
1151
|
+
// and refusals land on the turn's trace, so the chat surface can answer
|
|
1152
|
+
// "what did it actually run, and what did it decline?" from evidence.
|
|
1153
|
+
onTraceEvent: (event) => recordTraceEvent(chatTraceId, event),
|
|
1072
1154
|
deps: {
|
|
1073
1155
|
callModel: callModelWithTrace,
|
|
1074
1156
|
executeTool: async (name, args, ctx) => {
|
|
@@ -1129,7 +1211,55 @@ export class ChatCommand extends BaseCommand {
|
|
|
1129
1211
|
tools: result.successfulToolCalls ?? result.toolCalls,
|
|
1130
1212
|
unverifiedActionClaim: result.unverifiedActionClaim,
|
|
1131
1213
|
unfulfilledPromise: result.unfulfilledPromise,
|
|
1214
|
+
unverifiedEdit: result.unverifiedEdit,
|
|
1215
|
+
unverifiedEditClaim: result.unverifiedEditClaim,
|
|
1216
|
+
undeliveredArtifact: result.undeliveredArtifact,
|
|
1132
1217
|
}));
|
|
1218
|
+
// G3 — record what this turn actually did so the NEXT turn starts from it
|
|
1219
|
+
// (files changed, whether anything verified the work, and whether the user
|
|
1220
|
+
// reported a regression). Best-effort: the ledger must never break a turn.
|
|
1221
|
+
try {
|
|
1222
|
+
const activity = result.successfulToolCalls ?? [];
|
|
1223
|
+
const verified = activity.some((t) => t === 'run_terminal' || t === 'test' || t === 'browser' || t === 'run_cli');
|
|
1224
|
+
recordWorkingState(workingStatePath, {
|
|
1225
|
+
filesTouched: [...touchedFiles],
|
|
1226
|
+
toolsUsed: activity,
|
|
1227
|
+
verified,
|
|
1228
|
+
unverifiedEdit: result.unverifiedEdit === true,
|
|
1229
|
+
userMessage: message,
|
|
1230
|
+
});
|
|
1231
|
+
}
|
|
1232
|
+
catch {
|
|
1233
|
+
// Best-effort.
|
|
1234
|
+
}
|
|
1235
|
+
// G1 + G2 — surface the unverified-edit warning ON THE CONSOLE too (the
|
|
1236
|
+
// trace badge alone is invisible to a CLI/gateway user). Printed, never
|
|
1237
|
+
// appended to `content`, so the delivered answer, the gateway bubble and
|
|
1238
|
+
// the answer cache all stay clean.
|
|
1239
|
+
try {
|
|
1240
|
+
if (result.unverifiedEditClaim) {
|
|
1241
|
+
logger.warn(' ⚠️ This reply asserts a code change, but NOTHING verified it (no test / typecheck / build / browser run). Treat the change as UNVERIFIED.');
|
|
1242
|
+
}
|
|
1243
|
+
else if (result.unverifiedEdit) {
|
|
1244
|
+
logger.warn(' ⚠️ Files were changed this turn but no verification ran — the change is unverified.');
|
|
1245
|
+
}
|
|
1246
|
+
}
|
|
1247
|
+
catch {
|
|
1248
|
+
// Best-effort — a warning must never break the turn.
|
|
1249
|
+
}
|
|
1250
|
+
// G13b — the DELIVERABLE warning, on the console for the same reason: the
|
|
1251
|
+
// trace badge is invisible to a CLI user, and this failure is the one that
|
|
1252
|
+
// looks most like success. The reply reads as a finished 12-page story while
|
|
1253
|
+
// no file exists, so the reader is told plainly that the deliverable is
|
|
1254
|
+
// missing rather than left to discover it when the path is not there.
|
|
1255
|
+
try {
|
|
1256
|
+
if (result.undeliveredArtifact) {
|
|
1257
|
+
logger.warn(' ⚠️ This request asked for a written deliverable, but NO file was written this turn — the text above is the answer, not the artifact.');
|
|
1258
|
+
}
|
|
1259
|
+
}
|
|
1260
|
+
catch {
|
|
1261
|
+
// Best-effort.
|
|
1262
|
+
}
|
|
1133
1263
|
// Finalize the turn (cache + memory + registry telemetry).
|
|
1134
1264
|
// E3c: a generationFailed turn is NOT cached/persisted — the caller may
|
|
1135
1265
|
// fall back to the rule decision, and the failure text must not pollute
|
|
@@ -1175,6 +1305,7 @@ export class ChatCommand extends BaseCommand {
|
|
|
1175
1305
|
toolCalls: result.toolCalls,
|
|
1176
1306
|
unverifiedActionClaim: result.unverifiedActionClaim,
|
|
1177
1307
|
unfulfilledPromise: result.unfulfilledPromise,
|
|
1308
|
+
undeliveredArtifact: result.undeliveredArtifact,
|
|
1178
1309
|
};
|
|
1179
1310
|
}
|
|
1180
1311
|
/**
|
|
@@ -1287,6 +1418,9 @@ export class ChatCommand extends BaseCommand {
|
|
|
1287
1418
|
// telemetry name the model that failed, instead of "unknown".
|
|
1288
1419
|
if (effectiveModel)
|
|
1289
1420
|
session.model = effectiveModel;
|
|
1421
|
+
// G5 — and name the attempt itself, so the trace recorder reports the
|
|
1422
|
+
// real provider/model (post-failover) rather than the session default.
|
|
1423
|
+
this.lastAttempt = { provider: typ, model: effectiveModel };
|
|
1290
1424
|
const nativeKey = `${typ}|${effectiveModel ?? ''}`;
|
|
1291
1425
|
// Answer-quality resilience: a CONFUSED reply — the model talking about
|
|
1292
1426
|
// the tool contract (e.g. apologizing that "the provided example call
|
|
@@ -1738,6 +1872,15 @@ export class ChatCommand extends BaseCommand {
|
|
|
1738
1872
|
circuitBreakerStatus,
|
|
1739
1873
|
...(dispatch.taskIntentHint ? { taskIntentHint: dispatch.taskIntentHint } : {}),
|
|
1740
1874
|
}, this.configManager));
|
|
1875
|
+
// G5 — record the routing snapshot now (the winner + why), so the reasoning
|
|
1876
|
+
// trace for this turn can show the decision instead of a blank column.
|
|
1877
|
+
this.lastRouteSnapshot = {
|
|
1878
|
+
provider: decision.provider,
|
|
1879
|
+
model: decision.model,
|
|
1880
|
+
score: decision.score,
|
|
1881
|
+
complexity: String(decision.complexity),
|
|
1882
|
+
explanation: decision.explanation,
|
|
1883
|
+
};
|
|
1741
1884
|
// Walk the ranked candidates (winner first) and return the first available
|
|
1742
1885
|
// provider — never a provider that lacks a key or endpoint. Providers that
|
|
1743
1886
|
// already failed this message (excludeProviders) OR earlier in this session
|
|
@@ -1750,6 +1893,17 @@ export class ChatCommand extends BaseCommand {
|
|
|
1750
1893
|
return expiresAt !== undefined && expiresAt > exclusionTime;
|
|
1751
1894
|
};
|
|
1752
1895
|
// ── Cold-start probe (suggestion 3) ─────────────────────────────────────
|
|
1896
|
+
// Start the background warmup/exploration daemon on EVERY chat session, not
|
|
1897
|
+
// only a cold one (Models-page audit). It is idempotent (already-running is
|
|
1898
|
+
// a no-op) and unref'd, so it cannot hold the process open — and it is the
|
|
1899
|
+
// only thing that verifies models the router cannot see yet. Left cold-only,
|
|
1900
|
+
// the verified pool could only ever shrink as staleness retired models.
|
|
1901
|
+
try {
|
|
1902
|
+
startWarmupDaemon(this.configManager);
|
|
1903
|
+
}
|
|
1904
|
+
catch {
|
|
1905
|
+
// Best-effort — warmup must never break chat.
|
|
1906
|
+
}
|
|
1753
1907
|
// A fresh registry has zero verified models → routing would fall back to
|
|
1754
1908
|
// credential-based defaults and possibly fail into dead ends. Fire ONE
|
|
1755
1909
|
// background probe+spot-check so the registry learns from real API data.
|