agent-nuvira 3.3.9 → 3.3.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +6 -0
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +27 -17
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/config.d.ts +28 -0
- package/dist/cli/config.d.ts.map +1 -1
- package/dist/cli/config.js +155 -0
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/eval.d.ts +8 -0
- package/dist/cli/eval.d.ts.map +1 -1
- package/dist/cli/eval.js +64 -0
- package/dist/cli/eval.js.map +1 -1
- package/dist/cli/loop-executor.d.ts +13 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +22 -18
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/config/capability-mode.d.ts +122 -0
- package/dist/config/capability-mode.d.ts.map +1 -0
- package/dist/config/capability-mode.js +132 -0
- package/dist/config/capability-mode.js.map +1 -0
- package/dist/config/limits.d.ts +47 -0
- package/dist/config/limits.d.ts.map +1 -0
- package/dist/config/limits.js +64 -0
- package/dist/config/limits.js.map +1 -0
- package/dist/config/process-env.d.ts.map +1 -1
- package/dist/config/process-env.js +44 -0
- package/dist/config/process-env.js.map +1 -1
- package/dist/config/types.d.ts +77 -0
- package/dist/config/types.d.ts.map +1 -1
- package/dist/config/work-digest.d.ts +46 -0
- package/dist/config/work-digest.d.ts.map +1 -0
- package/dist/config/work-digest.js +80 -0
- package/dist/config/work-digest.js.map +1 -0
- package/dist/gateway/inbound-media.d.ts +18 -0
- package/dist/gateway/inbound-media.d.ts.map +1 -1
- package/dist/gateway/inbound-media.js +96 -2
- package/dist/gateway/inbound-media.js.map +1 -1
- package/dist/inference/anthropic-adapter.d.ts +6 -0
- package/dist/inference/anthropic-adapter.d.ts.map +1 -1
- package/dist/inference/anthropic-adapter.js +79 -8
- package/dist/inference/anthropic-adapter.js.map +1 -1
- package/dist/inference/gemini-adapter.d.ts +6 -0
- package/dist/inference/gemini-adapter.d.ts.map +1 -1
- package/dist/inference/gemini-adapter.js +73 -8
- package/dist/inference/gemini-adapter.js.map +1 -1
- package/dist/inference/groq-adapter.d.ts +3 -0
- package/dist/inference/groq-adapter.d.ts.map +1 -1
- package/dist/inference/groq-adapter.js +83 -40
- package/dist/inference/groq-adapter.js.map +1 -1
- package/dist/inference/interface.d.ts +7 -0
- package/dist/inference/interface.d.ts.map +1 -1
- package/dist/inference/model-probe.d.ts +17 -0
- package/dist/inference/model-probe.d.ts.map +1 -1
- package/dist/inference/model-probe.js +58 -1
- package/dist/inference/model-probe.js.map +1 -1
- package/dist/inference/model-validator.d.ts +16 -1
- package/dist/inference/model-validator.d.ts.map +1 -1
- package/dist/inference/model-validator.js +83 -2
- package/dist/inference/model-validator.js.map +1 -1
- package/dist/inference/nim-adapter.js +4 -4
- package/dist/inference/nim-adapter.js.map +1 -1
- package/dist/inference/openai-compat-adapter.d.ts +9 -0
- package/dist/inference/openai-compat-adapter.d.ts.map +1 -1
- package/dist/inference/openai-compat-adapter.js +101 -51
- package/dist/inference/openai-compat-adapter.js.map +1 -1
- package/dist/inference/openrouter-adapter.d.ts +3 -0
- package/dist/inference/openrouter-adapter.d.ts.map +1 -1
- package/dist/inference/openrouter-adapter.js +134 -57
- package/dist/inference/openrouter-adapter.js.map +1 -1
- package/dist/inference/reasoning-effort.d.ts +130 -0
- package/dist/inference/reasoning-effort.d.ts.map +1 -0
- package/dist/inference/reasoning-effort.js +237 -0
- package/dist/inference/reasoning-effort.js.map +1 -0
- package/dist/inference/route-resolver.d.ts +7 -0
- package/dist/inference/route-resolver.d.ts.map +1 -1
- package/dist/inference/route-resolver.js +2 -1
- package/dist/inference/route-resolver.js.map +1 -1
- package/dist/inference/sse.d.ts +6 -0
- package/dist/inference/sse.d.ts.map +1 -1
- package/dist/inference/sse.js +1 -0
- package/dist/inference/sse.js.map +1 -1
- package/dist/inference/tools.d.ts +13 -0
- package/dist/inference/tools.d.ts.map +1 -1
- package/dist/inference/tools.js +20 -2
- package/dist/inference/tools.js.map +1 -1
- package/dist/learning/autonomy-policy.d.ts +8 -0
- package/dist/learning/autonomy-policy.d.ts.map +1 -1
- package/dist/learning/autonomy-policy.js +34 -0
- package/dist/learning/autonomy-policy.js.map +1 -1
- package/dist/learning/capability-parity.d.ts +108 -0
- package/dist/learning/capability-parity.d.ts.map +1 -0
- package/dist/learning/capability-parity.js +154 -0
- package/dist/learning/capability-parity.js.map +1 -0
- package/dist/learning/cost-tracker.d.ts +28 -4
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +58 -8
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/eval-framework.d.ts +17 -0
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +63 -2
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/model-capability.d.ts +94 -0
- package/dist/learning/model-capability.d.ts.map +1 -0
- package/dist/learning/model-capability.js +172 -0
- package/dist/learning/model-capability.js.map +1 -0
- package/dist/learning/model-registry.d.ts +33 -0
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +79 -0
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/reasoning-trace.d.ts +1 -1
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resolve-options.d.ts.map +1 -1
- package/dist/learning/resolve-options.js +16 -3
- package/dist/learning/resolve-options.js.map +1 -1
- package/dist/learning/run-trace.d.ts +68 -0
- package/dist/learning/run-trace.d.ts.map +1 -1
- package/dist/learning/run-trace.js +101 -0
- package/dist/learning/run-trace.js.map +1 -1
- package/dist/tools/loop-skill-hint.d.ts +27 -0
- package/dist/tools/loop-skill-hint.d.ts.map +1 -1
- package/dist/tools/loop-skill-hint.js +189 -5
- package/dist/tools/loop-skill-hint.js.map +1 -1
- package/dist/tools/read-extract.d.ts +9 -3
- package/dist/tools/read-extract.d.ts.map +1 -1
- package/dist/tools/read-extract.js +70 -25
- package/dist/tools/read-extract.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +120 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +447 -5
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/web-dashboard/attachment-extract.d.ts +8 -1
- package/dist/web-dashboard/attachment-extract.d.ts.map +1 -1
- package/dist/web-dashboard/attachment-extract.js +14 -3
- package/dist/web-dashboard/attachment-extract.js.map +1 -1
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +160 -21
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +26 -0
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.js +44 -4
- package/dist/web-dashboard/workspace-guard.js.map +1 -1
- package/package.json +1 -1
- package/src/web-dashboard/public/assets/index-BdKf5Xw2.js +207 -0
- package/src/web-dashboard/public/assets/index-BdKf5Xw2.js.map +1 -0
- package/src/web-dashboard/public/assets/index-C9eBskc9.css +1 -0
- package/src/web-dashboard/public/index.html +2 -2
- package/src/web-dashboard/public/assets/index-Beportyl.js +0 -207
- package/src/web-dashboard/public/assets/index-Beportyl.js.map +0 -1
- package/src/web-dashboard/public/assets/index-gtQyg9nm.css +0 -1
package/dist/tools/tool-loop.js
CHANGED
|
@@ -33,11 +33,18 @@ import { stepDigest } from '../learning/step-checkpoint.js';
|
|
|
33
33
|
// process declared no fault (`NUVIRA_INJECT_FAULT`), so an ordinary turn is
|
|
34
34
|
// unaffected.
|
|
35
35
|
import { faultAt } from '../runtime/fault-injection.js';
|
|
36
|
-
import { detectPermissionSeeking, isAffirmativeReply, replyAsksTheReader, requestAuthorizesWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
|
|
36
|
+
import { detectPermissionSeeking, isAffirmativeReply, replyAsksTheReader, requestAuthorizesWrites, requestForbidsWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
|
|
37
37
|
import { envelopeFromPlan, envelopeFromRequest, getEnvelope, grantEnvelope, isEnvelopeKey, } from '../learning/intent-envelope.js';
|
|
38
|
-
import { detectProcessComplaint, isTraceKey, repeatNudge, runTraceFor, RunTrace, } from '../learning/run-trace.js';
|
|
38
|
+
import { detectProcessComplaint, isTraceKey, noProgressNudge, repeatNudge, repeatedFailureNudge, runTraceFor, RunTrace, } from '../learning/run-trace.js';
|
|
39
39
|
import { wantsAuthoredArtifact } from '../learning/deliverable-class.js';
|
|
40
40
|
import { normalizeFollowups } from './followup-utils.js';
|
|
41
|
+
// The capability mode (balanced | max) — `max` widens the loop's own reasoning
|
|
42
|
+
// budget as well as routing, so "cost is not a concern" means the agent may
|
|
43
|
+
// keep working through a long build instead of stopping at the default bound.
|
|
44
|
+
import { isMaxCapability } from '../config/capability-mode.js';
|
|
45
|
+
// The digest's scope (all | max | off) — a user-facing control over whether the
|
|
46
|
+
// compaction digest is injected in every mode, only under `max`, or never.
|
|
47
|
+
import { isWorkDigestEnabled } from '../config/work-digest.js';
|
|
41
48
|
import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, verificationNudgeFor, THINK_ONLY_ESCALATION, AUTHORIZED_WORK_NUDGE, deliverableNudge, } from './edit-verification.js';
|
|
42
49
|
import { effectiveToolJsonSchemas, coreToolJsonSchemas, isToolEnabled, toolsetForTool } from './toolsets.js';
|
|
43
50
|
import { deliverablesNamedIn, recordStepHandoff } from '../agents/step-handoff.js';
|
|
@@ -60,6 +67,22 @@ function mutatedPathOf(args) {
|
|
|
60
67
|
const p = a?.path ?? a?.file_path ?? a?.file;
|
|
61
68
|
return typeof p === 'string' && p.trim() ? p.trim() : undefined;
|
|
62
69
|
}
|
|
70
|
+
/**
|
|
71
|
+
* The ACTION signature of a failed call — what it tried to do, so the identical
|
|
72
|
+
* retry is recognizable. Prefers the shell `command`, then the file `path`,
|
|
73
|
+
* then the tool name (a call that named neither still repeats as the same tool).
|
|
74
|
+
* Used by the self-diagnosis gate (see RunTrace.repeatedFailure).
|
|
75
|
+
*/
|
|
76
|
+
function failureActionOf(call) {
|
|
77
|
+
const a = call.arguments;
|
|
78
|
+
const command = typeof a?.command === 'string' && a.command.trim() ? a.command.trim() : undefined;
|
|
79
|
+
if (command)
|
|
80
|
+
return command;
|
|
81
|
+
const path = mutatedPathOf(call.arguments);
|
|
82
|
+
if (path)
|
|
83
|
+
return path;
|
|
84
|
+
return call.name;
|
|
85
|
+
}
|
|
63
86
|
/**
|
|
64
87
|
* Append what a tool call actually did to the turn's provenance ledger.
|
|
65
88
|
*
|
|
@@ -396,10 +419,38 @@ const MAX_PARALLEL_READS = 4;
|
|
|
396
419
|
* were well able to do (see `THINK_ONLY_ESCALATION`).
|
|
397
420
|
*/
|
|
398
421
|
export const MAX_THINK_CONTINUES = 3;
|
|
422
|
+
/**
|
|
423
|
+
* Consecutive steps that RAN tools and had NO success before the loop asks the
|
|
424
|
+
* model to diagnose the stall. This generalises {@link RunTrace.repeatedFailure}
|
|
425
|
+
* (the identical action retried): a weak model also loops by substituting a
|
|
426
|
+
* different command each step, every one failing the same underlying way — none
|
|
427
|
+
* of them repeats exactly, so the same-action gate never fires.
|
|
428
|
+
*
|
|
429
|
+
* The threshold is deliberately ABOVE the handful of distinct checks ordinary
|
|
430
|
+
* iteration tries (`npm test` → `npm run build` → `npm run lint`), so a short run
|
|
431
|
+
* of unrelated failures is left alone; only a sustained all-fail streak is a stall.
|
|
432
|
+
*/
|
|
433
|
+
export const NO_PROGRESS_STALL_STEPS = 4;
|
|
434
|
+
/**
|
|
435
|
+
* How substantial a turn must be before the SELF-REVIEW gate may fire (see
|
|
436
|
+
* `selfReviewNudge`). Short, single-edit turns already end with an obvious
|
|
437
|
+
* result the model just looked at; a long turn is where scope drift hides, and
|
|
438
|
+
* where an extra pass earns its latency.
|
|
439
|
+
*/
|
|
440
|
+
export const SELF_REVIEW_MIN_STEPS = 6;
|
|
399
441
|
/** Default continuations granted per turn when the option is omitted. */
|
|
400
442
|
export const DEFAULT_MAX_CONTINUATIONS = 2;
|
|
401
443
|
/** Default extra steps granted per continuation. */
|
|
402
444
|
export const DEFAULT_CONTINUATION_STEPS = 8;
|
|
445
|
+
/**
|
|
446
|
+
* `max` capability mode — the loop's longest bounded reasoning budget. The
|
|
447
|
+
* defaults above are sized to keep an ordinary turn responsive; under `max` the
|
|
448
|
+
* user has said cost is not a concern, so the same session is allowed to run a
|
|
449
|
+
* long build / many-file edit to completion instead of stopping at the default
|
|
450
|
+
* bound. Still finite: the hard cap is `maxSteps + 4 * 16` extra steps.
|
|
451
|
+
*/
|
|
452
|
+
export const MAX_CAPABILITY_MAX_CONTINUATIONS = 4;
|
|
453
|
+
export const MAX_CAPABILITY_CONTINUATION_STEPS = 16;
|
|
403
454
|
/** Pause before re-attempting a failed step (lets a transient outage clear). */
|
|
404
455
|
export const CONTINUATION_DELAY_MS = 1_500;
|
|
405
456
|
/**
|
|
@@ -425,8 +476,12 @@ function sleep(ms) {
|
|
|
425
476
|
async function runToolLoopInner(opts, progress) {
|
|
426
477
|
const { messages, tools: toolNames, maxSteps = 16, context, deps } = opts;
|
|
427
478
|
// Bounded auto-continuation state (see ToolLoopOptions.maxContinuations).
|
|
428
|
-
|
|
429
|
-
|
|
479
|
+
// `max` capability mode raises the DEFAULT budget (an explicit option still
|
|
480
|
+
// wins, so callers that pin a bound keep it). Read once per turn from the
|
|
481
|
+
// config, exactly like routing, so a mode change applies to the next turn.
|
|
482
|
+
const maxCapability = isMaxCapability(context.configManager);
|
|
483
|
+
const maxContinuations = Math.max(0, opts.maxContinuations ?? (maxCapability ? MAX_CAPABILITY_MAX_CONTINUATIONS : DEFAULT_MAX_CONTINUATIONS));
|
|
484
|
+
const continuationSteps = Math.max(1, opts.continuationSteps ?? (maxCapability ? MAX_CAPABILITY_CONTINUATION_STEPS : DEFAULT_CONTINUATION_STEPS));
|
|
430
485
|
let continuations = 0;
|
|
431
486
|
// Bounded dangling-promise nudges spent this turn (see INTENT_PROMISE_RE).
|
|
432
487
|
let intentNudges = 0;
|
|
@@ -440,12 +495,26 @@ async function runToolLoopInner(opts, progress) {
|
|
|
440
495
|
// asks the model to PROCEED, this one asks it to stop REPEATING, and the two
|
|
441
496
|
// fire on independent evidence.
|
|
442
497
|
let repeatNudges = 0;
|
|
498
|
+
// The loop's SELF-DIAGNOSIS nudge: bounded once per turn, fired when the SAME
|
|
499
|
+
// action has failed more than once (see RunTrace.repeatedFailure). This is the
|
|
500
|
+
// capability that turns "retry the identical failing command 10 times" into
|
|
501
|
+
// "state the root cause and change the approach" — the live macOS-build turn
|
|
502
|
+
// re-issued `npx tauri build` again and again against the same missing Cargo.
|
|
503
|
+
let diagnosisNudges = 0;
|
|
504
|
+
// Stage 2 — consecutive tool-running steps with NO success (the generalized
|
|
505
|
+
// stall signal, see NO_PROGRESS_STALL_STEPS). Reset by any successful tool.
|
|
506
|
+
let failedToolSteps = 0;
|
|
507
|
+
// Stage 4 — bounded SELF-REVIEW nudge (see requireSelfReview / selfReviewNudge).
|
|
508
|
+
let selfReviewNudges = 0;
|
|
443
509
|
// Bounded think-only continuation (see THINK_ONLY_ESCALATION). Counts the
|
|
444
510
|
// consecutive reasoning-only steps so the loop cannot spin on them.
|
|
445
511
|
let thinkContinues = 0;
|
|
446
512
|
// G13b — bounded "the request asked for a file and none was written" nudges
|
|
447
513
|
// (see wantsAuthoredArtifact).
|
|
448
514
|
let deliverableNudges = 0;
|
|
515
|
+
// ZERO-ACTION — bounded "the request directed work and NOTHING was done"
|
|
516
|
+
// nudges (see requireAction / zeroActionGateApplies).
|
|
517
|
+
let actionNudges = 0;
|
|
449
518
|
/**
|
|
450
519
|
* G18 — the sink, wrapped so an observability failure can never become a
|
|
451
520
|
* turn failure (a recorder that throws is a bug in the instrument, not in
|
|
@@ -646,6 +715,27 @@ async function runToolLoopInner(opts, progress) {
|
|
|
646
715
|
// with an empty/bounded response (the "where is the essay?" bug).
|
|
647
716
|
let lastContent = '';
|
|
648
717
|
let bounded = false;
|
|
718
|
+
/**
|
|
719
|
+
* Should the SELF-REVIEW gate fire now? Returns the correction to inject, or
|
|
720
|
+
* null. Extracted so the two end-of-turn exits ask the question identically —
|
|
721
|
+
* nothing here is a function of which branch is ending.
|
|
722
|
+
*/
|
|
723
|
+
const selfReviewCorrection = () => {
|
|
724
|
+
if (selfReviewNudges >= 1)
|
|
725
|
+
return null;
|
|
726
|
+
if (opts.requireSelfReview === false)
|
|
727
|
+
return null;
|
|
728
|
+
if (steps < SELF_REVIEW_MIN_STEPS)
|
|
729
|
+
return null;
|
|
730
|
+
if (progress.mutatedPaths.length === 0)
|
|
731
|
+
return null;
|
|
732
|
+
// The verification gate owns "did you check it"; self-review only runs once
|
|
733
|
+
// that question is settled, so the two can never both fire in one turn.
|
|
734
|
+
const activity = assessEditActivity(progress.successfulToolCalls, progress.verificationEvidence, progress.mutatedPaths);
|
|
735
|
+
if (activity.needsVerification)
|
|
736
|
+
return null;
|
|
737
|
+
return selfReviewNudge(currentAsk(opts));
|
|
738
|
+
};
|
|
649
739
|
for (;;) {
|
|
650
740
|
// ── Bounded auto-continuation on the STEP BOUND ────────────────────────
|
|
651
741
|
// The model still wanted to act when the budget ran out (a long build, a
|
|
@@ -675,6 +765,19 @@ async function runToolLoopInner(opts, progress) {
|
|
|
675
765
|
if (trimmedResult.trimmed > 0) {
|
|
676
766
|
thread.length = 0;
|
|
677
767
|
thread.push(...trimmedResult.thread);
|
|
768
|
+
// Keep a memory of what the turn DID after its raw outputs are gone:
|
|
769
|
+
// without this, a long turn loses the VERDICTS (did the build pass?) and
|
|
770
|
+
// re-runs work it already finished. Deterministic and bounded — the same
|
|
771
|
+
// philosophy as trimThreadBudget itself (facts, no summarizer, no drift).
|
|
772
|
+
if (isWorkDigestEnabled(context.configManager)) {
|
|
773
|
+
const digest = buildWorkDigest({
|
|
774
|
+
successfulTools: progress.successfulToolCalls,
|
|
775
|
+
mutatedPaths: progress.mutatedPaths,
|
|
776
|
+
executedActions: progress.executedActions,
|
|
777
|
+
});
|
|
778
|
+
if (digest)
|
|
779
|
+
upsertWorkDigest(thread, digest);
|
|
780
|
+
}
|
|
678
781
|
deps.onEvent?.(` ✂️ ${trimmedResult.trimmed} old tool result(s) trimmed to fit the ${Math.round(budgetChars / 1000)}K-char context budget.`);
|
|
679
782
|
}
|
|
680
783
|
}
|
|
@@ -999,6 +1102,32 @@ async function runToolLoopInner(opts, progress) {
|
|
|
999
1102
|
thread.push({ role: 'user', content: repeatNudge(closing, runTrace.priorAnswer(closing)) });
|
|
1000
1103
|
continue;
|
|
1001
1104
|
}
|
|
1105
|
+
// ── ZERO-ACTION gate (bounded, once) ────────────────────────────────
|
|
1106
|
+
// The request DIRECTED work on the workspace and the turn is ending
|
|
1107
|
+
// without having run a single tool: not a dropped promise, not a
|
|
1108
|
+
// permission question, not an authored file — just prose where the work
|
|
1109
|
+
// should be. The gates above all key on a positive shape in the reply and
|
|
1110
|
+
// so miss a plain non-answer; this one keys on the REQUEST and the
|
|
1111
|
+
// ABSENCE of action, which is exactly the shape that ended a fully
|
|
1112
|
+
// specified four-part coding ask with nothing done. Bounded once; the
|
|
1113
|
+
// residual is reported by `noActionTaken` below, whether or not the nudge
|
|
1114
|
+
// is enabled.
|
|
1115
|
+
if (actionNudges < 1 &&
|
|
1116
|
+
zeroActionGateApplies(opts, progress, schemas.length, requestText, authorization.authorized)) {
|
|
1117
|
+
actionNudges += 1;
|
|
1118
|
+
stepLimit += 1;
|
|
1119
|
+
deps.onEvent?.(' 🛠️ The request asked for work and nothing was done — telling the model to do it now.');
|
|
1120
|
+
traceEvent({
|
|
1121
|
+
kind: 'gate',
|
|
1122
|
+
gate: 'action',
|
|
1123
|
+
summary: 'the request directed work on the workspace and the turn performed none — one bounded nudge to carry it out',
|
|
1124
|
+
});
|
|
1125
|
+
// The non-answer must not become the delivered answer either way.
|
|
1126
|
+
lastContent = '';
|
|
1127
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
1128
|
+
thread.push({ role: 'user', content: zeroActionNudge(currentAsk(opts)) });
|
|
1129
|
+
continue;
|
|
1130
|
+
}
|
|
1002
1131
|
// S1 (both exits): the MOST SUBSTANTIVE content seen wins here too —
|
|
1003
1132
|
// a short closing step ("Sent it to her! ✅") with no tool calls must
|
|
1004
1133
|
// not clobber the deliverable (poem/essay) the model composed in an
|
|
@@ -1055,6 +1184,22 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1055
1184
|
});
|
|
1056
1185
|
continue;
|
|
1057
1186
|
}
|
|
1187
|
+
// SELF-REVIEW (no-tools exit) — a substantial, already-verified turn is
|
|
1188
|
+
// ending; spend ONE bounded pass to check the result against the ask.
|
|
1189
|
+
const review = selfReviewCorrection();
|
|
1190
|
+
if (review) {
|
|
1191
|
+
selfReviewNudges += 1;
|
|
1192
|
+
stepLimit += 1;
|
|
1193
|
+
deps.onEvent?.(' 🧭 Substantial turn ending — asking the model to review the result against the original ask.');
|
|
1194
|
+
traceEvent({
|
|
1195
|
+
kind: 'gate',
|
|
1196
|
+
gate: 'self-review',
|
|
1197
|
+
summary: 'a substantial turn that changed files reached its end — one bounded nudge to check the result against the original ask',
|
|
1198
|
+
});
|
|
1199
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
1200
|
+
thread.push({ role: 'user', content: review });
|
|
1201
|
+
continue;
|
|
1202
|
+
}
|
|
1058
1203
|
return {
|
|
1059
1204
|
content: response.content.length >= lastContent.length ? response.content : lastContent,
|
|
1060
1205
|
followups,
|
|
@@ -1323,6 +1468,7 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1323
1468
|
// What each call actually delivered (post hint/tip decoration), kept in
|
|
1324
1469
|
// call order for the endsAgentStep exit below.
|
|
1325
1470
|
const delivered = new Array(plans.length).fill('');
|
|
1471
|
+
let stepAnySuccess = false;
|
|
1326
1472
|
for (let i = 0; i < plans.length; i += 1) {
|
|
1327
1473
|
const call = plans[i].call;
|
|
1328
1474
|
const rawResult = executed[i];
|
|
@@ -1342,8 +1488,10 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1342
1488
|
const ranOk = refusal === null &&
|
|
1343
1489
|
!rawResult.startsWith('Error:') &&
|
|
1344
1490
|
(!DELIVERY_TOOL_NAMES.has(call.name) || deliveryResultSucceeded(rawResult));
|
|
1345
|
-
if (ranOk)
|
|
1491
|
+
if (ranOk) {
|
|
1346
1492
|
progress.successfulToolCalls.push(call.name);
|
|
1493
|
+
stepAnySuccess = true;
|
|
1494
|
+
}
|
|
1347
1495
|
if (call.name === 'gateway_send' && deliveryResultSucceeded(rawResult)) {
|
|
1348
1496
|
progress.deliveryConfirmed = true;
|
|
1349
1497
|
}
|
|
@@ -1367,6 +1515,17 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1367
1515
|
}
|
|
1368
1516
|
}
|
|
1369
1517
|
else if (refusal !== null || rawResult.startsWith('Error:')) {
|
|
1518
|
+
// SELF-DIAGNOSIS — record an action that RAN and FAILED (a non-zero
|
|
1519
|
+
// exit, a timeout, a tool error), keyed by the action itself (the
|
|
1520
|
+
// command / path). The gate below fires when the SAME action repeats:
|
|
1521
|
+
// a refusal was already remembered across runs (step-handoff), but a
|
|
1522
|
+
// command that RUNS and FAILS was not, so the live macOS-build turn
|
|
1523
|
+
// re-issued `npx tauri build` ~10 times against the same missing Cargo
|
|
1524
|
+
// and never once stopped to diagnose. Refusals are the other record
|
|
1525
|
+
// below; this one is for the calls that actually executed.
|
|
1526
|
+
if (refusal === null) {
|
|
1527
|
+
runTrace.recordFailure(call.name, failureActionOf(call), rawResult);
|
|
1528
|
+
}
|
|
1370
1529
|
// Stage 2 — the other half of noticing a loop: the SAME refusal, twice.
|
|
1371
1530
|
// The run records it with its reason, so a later step (or a self-report)
|
|
1372
1531
|
// can see "blocked twice, identically" instead of rediscovering it — and
|
|
@@ -1443,6 +1602,52 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1443
1602
|
});
|
|
1444
1603
|
thread.push({ role: 'tool', toolCallId: call.id, content: resultText });
|
|
1445
1604
|
}
|
|
1605
|
+
// Track the generalized stall: a step that RAN tools but succeeded at none
|
|
1606
|
+
// of them extends the streak; any success clears it. A text-only step (no
|
|
1607
|
+
// tools) leaves it as it was — the model is thinking, not failing.
|
|
1608
|
+
if (plans.length > 0)
|
|
1609
|
+
failedToolSteps = stepAnySuccess ? 0 : failedToolSteps + 1;
|
|
1610
|
+
// ── SELF-DIAGNOSIS nudge (bounded, once) ──────────────────────────────
|
|
1611
|
+
// The loop's missing introspection: when the SAME action has failed more
|
|
1612
|
+
// than once, re-issuing it is a loop, not progress. This is the exact
|
|
1613
|
+
// shape of the live macOS-build turn — `npx tauri build` was retried ~10
|
|
1614
|
+
// times against the same missing Rust/Cargo, each attempt rediscovering
|
|
1615
|
+
// the same wall, and the run never stopped to ask WHY. Unlike the
|
|
1616
|
+
// end-of-turn gates (permission/repeat/deliverable), which fire only when
|
|
1617
|
+
// the model stops calling tools, this one must fire MID-TURN while the
|
|
1618
|
+
// model is still looping on the failing call — so it is injected here,
|
|
1619
|
+
// right after the step's results are in the thread, and the next model
|
|
1620
|
+
// step sees it. Bounded once per turn and keyed on the action so the same
|
|
1621
|
+
// repeat cannot trigger it twice.
|
|
1622
|
+
if (diagnosisNudges < 1) {
|
|
1623
|
+
const repeated = runTrace.repeatedFailure();
|
|
1624
|
+
if (repeated && repeated.times >= 2) {
|
|
1625
|
+
diagnosisNudges += 1;
|
|
1626
|
+
stepLimit += 1;
|
|
1627
|
+
deps.onEvent?.(` 🩺 Same action failed ${repeated.times}× — asking the model to diagnose the cause instead of retrying.`);
|
|
1628
|
+
traceEvent({
|
|
1629
|
+
kind: 'gate',
|
|
1630
|
+
gate: 'diagnosis',
|
|
1631
|
+
summary: `the same action failed ${repeated.times} times (${repeated.tool}: ${repeated.action.slice(0, 120)}) — ` +
|
|
1632
|
+
'one bounded nudge to diagnose the root cause and change the approach',
|
|
1633
|
+
});
|
|
1634
|
+
thread.push({ role: 'user', content: repeatedFailureNudge(repeated) });
|
|
1635
|
+
}
|
|
1636
|
+
else if (failedToolSteps >= NO_PROGRESS_STALL_STEPS) {
|
|
1637
|
+
// Stage 2 — no SINGLE action repeated, but the last N steps each ran
|
|
1638
|
+
// tools and none succeeded. Same diagnosis demanded, different evidence.
|
|
1639
|
+
diagnosisNudges += 1;
|
|
1640
|
+
stepLimit += 1;
|
|
1641
|
+
deps.onEvent?.(` 🩺 ${failedToolSteps} steps with no successful action — asking the model to diagnose the stall.`);
|
|
1642
|
+
traceEvent({
|
|
1643
|
+
kind: 'gate',
|
|
1644
|
+
gate: 'diagnosis',
|
|
1645
|
+
summary: `${failedToolSteps} consecutive steps ran tools and none succeeded — ` +
|
|
1646
|
+
'one bounded nudge to diagnose the shared cause and change the approach',
|
|
1647
|
+
});
|
|
1648
|
+
thread.push({ role: 'user', content: noProgressNudge(runTrace.recentFailures(3), failedToolSteps) });
|
|
1649
|
+
}
|
|
1650
|
+
}
|
|
1446
1651
|
// Tiered exposure: after EVERY executed tool call, union any newly
|
|
1447
1652
|
// loaded toolset schemas into the live set so the NEXT model step can
|
|
1448
1653
|
// call them natively (tool_search load → loadedExtraTools → here).
|
|
@@ -1571,6 +1776,40 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1571
1776
|
thread.push({ role: 'user', content: deliverableNudge(authorization.requestedPath) });
|
|
1572
1777
|
continue;
|
|
1573
1778
|
}
|
|
1779
|
+
// ZERO-ACTION gate (concluding exit) — the same bounded pass as the
|
|
1780
|
+
// no-tools exit, for a turn that concluded (suggest_followups) without
|
|
1781
|
+
// ever touching the workspace a directed request asked it to change.
|
|
1782
|
+
if (actionNudges < 1 &&
|
|
1783
|
+
zeroActionGateApplies(opts, progress, schemas.length, requestText, authorization.authorized)) {
|
|
1784
|
+
actionNudges += 1;
|
|
1785
|
+
stepLimit += 1;
|
|
1786
|
+
deps.onEvent?.(' 🛠️ The request asked for work and nothing was done — telling the model to do it now.');
|
|
1787
|
+
traceEvent({
|
|
1788
|
+
kind: 'gate',
|
|
1789
|
+
gate: 'action',
|
|
1790
|
+
summary: 'the request directed work on the workspace and the turn performed none — one bounded nudge to carry it out',
|
|
1791
|
+
});
|
|
1792
|
+
lastContent = '';
|
|
1793
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
1794
|
+
thread.push({ role: 'user', content: zeroActionNudge(currentAsk(opts)) });
|
|
1795
|
+
continue;
|
|
1796
|
+
}
|
|
1797
|
+
// SELF-REVIEW (concluding exit) — same bounded pass, placed AFTER the
|
|
1798
|
+
// verification/deliverable gates so those settle their questions first.
|
|
1799
|
+
const review = selfReviewCorrection();
|
|
1800
|
+
if (review) {
|
|
1801
|
+
selfReviewNudges += 1;
|
|
1802
|
+
stepLimit += 1;
|
|
1803
|
+
deps.onEvent?.(' 🧭 Substantial turn ending — asking the model to review the result against the original ask.');
|
|
1804
|
+
traceEvent({
|
|
1805
|
+
kind: 'gate',
|
|
1806
|
+
gate: 'self-review',
|
|
1807
|
+
summary: 'a substantial turn that changed files reached its end — one bounded nudge to check the result against the original ask',
|
|
1808
|
+
});
|
|
1809
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
1810
|
+
thread.push({ role: 'user', content: review });
|
|
1811
|
+
continue;
|
|
1812
|
+
}
|
|
1574
1813
|
const content = response.content.length >= lastContent.length ? response.content : lastContent;
|
|
1575
1814
|
return {
|
|
1576
1815
|
content,
|
|
@@ -1921,9 +2160,23 @@ export async function runToolLoop(opts) {
|
|
|
1921
2160
|
if (!result.cancelled &&
|
|
1922
2161
|
progress.mutatedPaths.length === 0 &&
|
|
1923
2162
|
!replyAsksTheReader(result.content) &&
|
|
2163
|
+
// A request that forbade writes cannot have "failed to deliver" a file.
|
|
2164
|
+
!requestForbidsWrites(lastUserText(opts.messages)) &&
|
|
1924
2165
|
wantsAuthoredArtifact(lastUserText(opts.messages))) {
|
|
1925
2166
|
result.undeliveredArtifact = true;
|
|
1926
2167
|
}
|
|
2168
|
+
// ZERO-ACTION honesty — a request that DIRECTED work on the workspace was
|
|
2169
|
+
// answered with nothing at all: no tool succeeded, nothing was written. The
|
|
2170
|
+
// flag is a function of what the turn DID, never of configuration, so a
|
|
2171
|
+
// caller can never read "completed" from a turn whose request it never
|
|
2172
|
+
// touched. Authored-artifact asks are left to `undeliveredArtifact` above so
|
|
2173
|
+
// one turn is never reported under two names.
|
|
2174
|
+
const askText = lastUserText(opts.messages);
|
|
2175
|
+
if (!hasProductiveAction(progress) &&
|
|
2176
|
+
progress.mutatedPaths.length === 0 &&
|
|
2177
|
+
requestRequiresWorkspaceAction(askText, requestAuthorizesWrites(askText).authorized)) {
|
|
2178
|
+
result.noActionTaken = true;
|
|
2179
|
+
}
|
|
1927
2180
|
}
|
|
1928
2181
|
return result;
|
|
1929
2182
|
}
|
|
@@ -1961,8 +2214,109 @@ function deliverableGateApplies(opts, content, progress, schemaCount, requestTex
|
|
|
1961
2214
|
return false;
|
|
1962
2215
|
if (replyAsksTheReader(content))
|
|
1963
2216
|
return false;
|
|
2217
|
+
// A negative instruction outranks every positive signal. "Do not write any
|
|
2218
|
+
// files — answer in chat" contains a creation verb and a file-shaped noun, so
|
|
2219
|
+
// without this the gate read it as an authored-artifact ask and wrote a file
|
|
2220
|
+
// against the user's explicit instruction.
|
|
2221
|
+
if (requestForbidsWrites(requestText))
|
|
2222
|
+
return false;
|
|
1964
2223
|
return wantsAuthoredArtifact(requestText);
|
|
1965
2224
|
}
|
|
2225
|
+
/**
|
|
2226
|
+
* Tools that do not count as having DONE anything on their own. `suggest_followups`
|
|
2227
|
+
* concludes a turn; it produces no work, so a turn that only called it has still
|
|
2228
|
+
* performed nothing the request asked for. Every other successful tool counts as
|
|
2229
|
+
* an action (a read is an action), which keeps the zero-action gate conservative.
|
|
2230
|
+
*/
|
|
2231
|
+
/**
|
|
2232
|
+
* A token that NAMES a file (a real source/config extension). The strongest
|
|
2233
|
+
* signal that the request is about the WORKSPACE, not a chat answer.
|
|
2234
|
+
*/
|
|
2235
|
+
const REQUEST_FILE_TOKEN_RE = /\b[\w./~-]+\.(?:js|mjs|cjs|ts|tsx|jsx|py|rb|go|rs|java|kt|cs|cpp|cxx|cc|c|h|hpp|json|ya?ml|toml|ini|cfg|conf|md|markdown|txt|csv|tsv|html?|css|scss|sass|less|sql|sh|bash|zsh|fish|env|lock|xml|gradle|properties|vue|svelte|php|lua|pl|swift|dart|scala)\b/i;
|
|
2236
|
+
/** A verb that asks for a change to the workspace. */
|
|
2237
|
+
const WORK_EDIT_VERB_RE = /\b(?:fix|repair|refactor|edit|modify|updat|chang|add|implement|creat|writ|delet|remov|renam|rewrit|migrat|correct|debug|patch|tweak|scaffold)\w*/i;
|
|
2238
|
+
/** A code/workspace noun that pairs with the verb above. */
|
|
2239
|
+
const CODE_NOUN_RE = /\b(?:file|files|function|functions|method|methods|class|classes|module|modules|script|scripts|component|components|test|tests|suite|api|endpoint|endpoints|route|routes|schema|schemas|migration|migrations|package|dependency|dependencies|import|imports|config|configuration|type|types|interface|interfaces|bug|bugs|repo|repository|codebase|project|source)\b/i;
|
|
2240
|
+
/**
|
|
2241
|
+
* Does this request DIRECT work on the workspace — the precondition for the
|
|
2242
|
+
* ZERO-ACTION gate?
|
|
2243
|
+
*
|
|
2244
|
+
* Deliberately narrower than "the request authorizes writes": the gate spends a
|
|
2245
|
+
* model step, so it must only fire on an ask that genuinely needs the tools.
|
|
2246
|
+
* `requestAuthorizesWrites` is true for a prose deliverable ("draft an
|
|
2247
|
+
* itinerary") or a review ("assess this project"), which are correctly answered
|
|
2248
|
+
* in chat — nudging those would turn one turn into two for no reason (found by
|
|
2249
|
+
* the existing suite). The workspace signal is therefore explicit: the request
|
|
2250
|
+
* either names a FILE, or pairs an edit verb with a code noun.
|
|
2251
|
+
*/
|
|
2252
|
+
function requestRequiresWorkspaceAction(requestText, authorized) {
|
|
2253
|
+
const text = (requestText || '').trim();
|
|
2254
|
+
if (!text)
|
|
2255
|
+
return false;
|
|
2256
|
+
if (!authorized)
|
|
2257
|
+
return false;
|
|
2258
|
+
if (requestForbidsWrites(text))
|
|
2259
|
+
return false;
|
|
2260
|
+
if (wantsAuthoredArtifact(text))
|
|
2261
|
+
return false;
|
|
2262
|
+
if (REQUEST_FILE_TOKEN_RE.test(text))
|
|
2263
|
+
return true;
|
|
2264
|
+
return WORK_EDIT_VERB_RE.test(text) && CODE_NOUN_RE.test(text);
|
|
2265
|
+
}
|
|
2266
|
+
const NON_PRODUCTIVE_TOOLS = new Set(['suggest_followups']);
|
|
2267
|
+
/** True when the turn performed at least one action beyond merely concluding. */
|
|
2268
|
+
function hasProductiveAction(progress) {
|
|
2269
|
+
return progress.successfulToolCalls.some((name) => !NON_PRODUCTIVE_TOOLS.has(name));
|
|
2270
|
+
}
|
|
2271
|
+
/**
|
|
2272
|
+
* Does the ZERO-ACTION gate apply to this turn?
|
|
2273
|
+
*
|
|
2274
|
+
* Every term is an independent, checkable fact — the conjunction is what keeps
|
|
2275
|
+
* the gate from firing on turns that are legitimately answer-only:
|
|
2276
|
+
*
|
|
2277
|
+
* - the request AUTHORIZES writes (`authorized`) — a create/maintenance ask,
|
|
2278
|
+
* not a question (`requestAuthorizesWrites` already vetoes pure questions);
|
|
2279
|
+
* - NO tool call succeeded this turn (a turn that gathered context and then
|
|
2280
|
+
* answered is out of scope — it did something, even if it then stalled);
|
|
2281
|
+
* - NOTHING was written (`mutatedPaths` is the loop's own proof a write
|
|
2282
|
+
* landed, so a heredoc through `run_terminal` is not nudged);
|
|
2283
|
+
* - tools are actually available (a caller that exposed none cannot comply);
|
|
2284
|
+
* - the request did NOT forbid writes (a negative instruction outranks every
|
|
2285
|
+
* positive signal — see `requestForbidsWrites`);
|
|
2286
|
+
* - it is NOT an authored-artifact ask: `wantsAuthoredArtifact` requests are
|
|
2287
|
+
* the DELIVERABLE gate's job, with their own narrower message, and running
|
|
2288
|
+
* both would spend two nudges on one ask.
|
|
2289
|
+
*/
|
|
2290
|
+
function zeroActionGateApplies(opts, progress, schemaCount, requestText, authorized) {
|
|
2291
|
+
if (opts.requireAction === false)
|
|
2292
|
+
return false;
|
|
2293
|
+
if (schemaCount === 0)
|
|
2294
|
+
return false;
|
|
2295
|
+
if (hasProductiveAction(progress))
|
|
2296
|
+
return false;
|
|
2297
|
+
if (progress.mutatedPaths.length > 0)
|
|
2298
|
+
return false;
|
|
2299
|
+
return requestRequiresWorkspaceAction(requestText, authorized);
|
|
2300
|
+
}
|
|
2301
|
+
/**
|
|
2302
|
+
* The bounded ZERO-ACTION correction — the loop telling the model, in one step,
|
|
2303
|
+
* that a directed request has not been touched yet.
|
|
2304
|
+
*
|
|
2305
|
+
* Names the ask (so the model cannot claim it did not know what was wanted),
|
|
2306
|
+
* states plainly that nothing ran and nothing changed, and closes the two escape
|
|
2307
|
+
* hatches that produced the observed non-actions: do not re-ask a request that is
|
|
2308
|
+
* already specified, and do not answer a work request with a plan, an apology or
|
|
2309
|
+
* a question. It still leaves a REAL blocker as a legitimate way out — the goal
|
|
2310
|
+
* is the work, not compliance theatre.
|
|
2311
|
+
*/
|
|
2312
|
+
export function zeroActionNudge(ask) {
|
|
2313
|
+
const quoted = (ask || '').trim().replace(/\s+/g, ' ').slice(0, 300);
|
|
2314
|
+
return ('Nothing has been done yet: you have not called a single tool this turn, so no file was changed and nothing was checked.' +
|
|
2315
|
+
(quoted ? ` The request already asked for this work: "${quoted}".` : '') +
|
|
2316
|
+
'\nYou have the tools to do it — read what you need, make the change, and run the check NOW.' +
|
|
2317
|
+
' Do NOT reply with a plan, a summary of what you would do, an apology, or a request for the user to restate or confirm' +
|
|
2318
|
+
' a request that is already complete. If something genuinely blocks you, say exactly what it is and why; otherwise do the work.');
|
|
2319
|
+
}
|
|
1966
2320
|
/**
|
|
1967
2321
|
* The text of the LAST user message in the thread — what the user is actually
|
|
1968
2322
|
* asking for right now, as opposed to the history above it. Used to derive
|
|
@@ -2148,4 +2502,92 @@ export function trimThreadBudget(thread, maxChars = DEFAULT_THREAD_BUDGET_CHARS)
|
|
|
2148
2502
|
}
|
|
2149
2503
|
return { thread: out, trimmed };
|
|
2150
2504
|
}
|
|
2505
|
+
// ─── Within-turn work digest (the memory a trimmed thread keeps) ────────────
|
|
2506
|
+
// `trimThreadBudget` keeps the first 500 chars of each old tool result, but past
|
|
2507
|
+
// that the model loses the VERDICT of what it ran and can re-run finished work.
|
|
2508
|
+
// `working-state` solves this ACROSS turns; this solves it WITHIN one. It is
|
|
2509
|
+
// deliberately deterministic and LLM-free (facts: what changed, what commands
|
|
2510
|
+
// ran and whether they passed, which tools were used) — no summarizer, no
|
|
2511
|
+
// latency, no drift, exactly like the ledger and the budget it complements.
|
|
2512
|
+
/** Marker prefix identifying the loop's within-turn work digest message. */
|
|
2513
|
+
export const WORK_DIGEST_MARKER = '[work digest — your actions so far this turn]';
|
|
2514
|
+
/**
|
|
2515
|
+
* Format the turn's actions so far as a bounded, model-readable digest. Returns
|
|
2516
|
+
* '' when there is nothing worth saying, so a pristine turn adds no noise.
|
|
2517
|
+
*/
|
|
2518
|
+
export function buildWorkDigest(input) {
|
|
2519
|
+
const lines = [];
|
|
2520
|
+
const changed = [...new Set(input.mutatedPaths.filter(Boolean))];
|
|
2521
|
+
if (changed.length > 0) {
|
|
2522
|
+
const shown = changed.slice(-12);
|
|
2523
|
+
lines.push(`• Files changed (${changed.length}): ${shown.join(', ')}${changed.length > shown.length ? ', …' : ''}`);
|
|
2524
|
+
}
|
|
2525
|
+
// Commands with their verdict, newest first, deduped by command+verdict — the
|
|
2526
|
+
// single most valuable thing to retain (did the build/test pass?).
|
|
2527
|
+
const cmds = input.executedActions.filter((a) => typeof a.command === 'string' && a.command.trim());
|
|
2528
|
+
if (cmds.length > 0) {
|
|
2529
|
+
const seen = new Set();
|
|
2530
|
+
const shown = [];
|
|
2531
|
+
for (let i = cmds.length - 1; i >= 0 && shown.length < 8; i -= 1) {
|
|
2532
|
+
const a = cmds[i];
|
|
2533
|
+
const key = `${a.ok ? 'ok' : 'fail'}:${a.command}`;
|
|
2534
|
+
if (seen.has(key))
|
|
2535
|
+
continue;
|
|
2536
|
+
seen.add(key);
|
|
2537
|
+
shown.unshift(`${a.ok ? '✅' : '❌'} ${a.command}`);
|
|
2538
|
+
}
|
|
2539
|
+
lines.push('• Commands run:');
|
|
2540
|
+
for (const s of shown)
|
|
2541
|
+
lines.push(` ${s}`);
|
|
2542
|
+
}
|
|
2543
|
+
const tools = [...new Set(input.successfulTools)];
|
|
2544
|
+
if (tools.length > 0)
|
|
2545
|
+
lines.push(`• Tools used: ${tools.join(', ')}`);
|
|
2546
|
+
if (lines.length === 0)
|
|
2547
|
+
return '';
|
|
2548
|
+
return `${WORK_DIGEST_MARKER}\n${lines.join('\n')}`;
|
|
2549
|
+
}
|
|
2550
|
+
/**
|
|
2551
|
+
* Insert or refresh the single work-digest message. Idempotent: a digest already
|
|
2552
|
+
* in the thread is UPDATED in place, never stacked, so repeated compaction in a
|
|
2553
|
+
* long turn keeps one current digest rather than accumulating stale ones. Placed
|
|
2554
|
+
* just after the system prompt + first user message so it sits with the ask and
|
|
2555
|
+
* survives the NEXT compaction (the tail is never trimmed).
|
|
2556
|
+
*/
|
|
2557
|
+
export function upsertWorkDigest(thread, digest) {
|
|
2558
|
+
const existing = thread.findIndex((m) => m.content.startsWith(WORK_DIGEST_MARKER));
|
|
2559
|
+
if (existing !== -1) {
|
|
2560
|
+
thread[existing] = { ...thread[existing], content: digest };
|
|
2561
|
+
return;
|
|
2562
|
+
}
|
|
2563
|
+
let at = 0;
|
|
2564
|
+
while (at < thread.length && thread[at].role === 'system')
|
|
2565
|
+
at += 1;
|
|
2566
|
+
const firstUser = thread.findIndex((m) => m.role === 'user');
|
|
2567
|
+
const insertAt = Math.min(thread.length, firstUser === -1 ? at : Math.max(at, firstUser + 1));
|
|
2568
|
+
thread.splice(insertAt, 0, { role: 'user', content: digest });
|
|
2569
|
+
}
|
|
2570
|
+
/**
|
|
2571
|
+
* The bounded SELF-REVIEW correction — the loop's stand-in for a reviewer who
|
|
2572
|
+
* asks "is this actually what was asked for?".
|
|
2573
|
+
*
|
|
2574
|
+
* Fired once, only for a substantial turn that changed files and has already
|
|
2575
|
+
* been verified (the verification gate owns "did you check"). Its job is
|
|
2576
|
+
* orthogonal: catch a result that is verified but does not satisfy the WHOLE
|
|
2577
|
+
* original ask. It demands evidence from THIS turn and forbids padding the
|
|
2578
|
+
* answer with more prose — the failure mode it targets is a confident, verified
|
|
2579
|
+
* answer to a slightly wrong question.
|
|
2580
|
+
*/
|
|
2581
|
+
export function selfReviewNudge(ask) {
|
|
2582
|
+
const quoted = (ask || '').trim().replace(/\s+/g, ' ').slice(0, 300);
|
|
2583
|
+
return ('Before you finish, REVIEW your result against the ORIGINAL request' +
|
|
2584
|
+
(quoted ? `: "${quoted}".` : '.') +
|
|
2585
|
+
'\nAnswer these to yourself in one short pass, and fix anything that is not true:' +
|
|
2586
|
+
'\n 1. Does what you produced satisfy EVERY part of that request — not just the part you found easiest?' +
|
|
2587
|
+
'\n 2. Is each claim in your answer backed by a tool result from THIS turn — or is it an assumption you did not check?' +
|
|
2588
|
+
'\n 3. Is anything the request asked for still missing, half-done, or done for the wrong target?' +
|
|
2589
|
+
'\nIf everything is satisfied and evidenced, reply with a brief confirmation and stop.' +
|
|
2590
|
+
' If something is missing, do it NOW — or state plainly and specifically what is not done and why.' +
|
|
2591
|
+
' Do NOT restate the work in more words — verify it.');
|
|
2592
|
+
}
|
|
2151
2593
|
//# sourceMappingURL=tool-loop.js.map
|