agent-nuvira 3.3.11 → 3.3.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -1
- package/dist/agents/agents/reasoner.d.ts.map +1 -1
- package/dist/agents/agents/reasoner.js +13 -0
- package/dist/agents/agents/reasoner.js.map +1 -1
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +56 -20
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/cli/chat.d.ts +36 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +456 -43
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts.map +1 -1
- package/dist/cli/cli-program.js +8 -0
- package/dist/cli/cli-program.js.map +1 -1
- package/dist/cli/config.d.ts.map +1 -1
- package/dist/cli/config.js +35 -4
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +14 -0
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/knowledge.d.ts +23 -0
- package/dist/cli/knowledge.d.ts.map +1 -0
- package/dist/cli/knowledge.js +132 -0
- package/dist/cli/knowledge.js.map +1 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +57 -13
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/omniroute-control.d.ts +88 -0
- package/dist/cli/omniroute-control.d.ts.map +1 -0
- package/dist/cli/omniroute-control.js +179 -0
- package/dist/cli/omniroute-control.js.map +1 -0
- package/dist/cli/omniroute.d.ts +21 -0
- package/dist/cli/omniroute.d.ts.map +1 -0
- package/dist/cli/omniroute.js +67 -0
- package/dist/cli/omniroute.js.map +1 -0
- package/dist/cli/tool-install-prompt.d.ts +58 -0
- package/dist/cli/tool-install-prompt.d.ts.map +1 -1
- package/dist/cli/tool-install-prompt.js +137 -0
- package/dist/cli/tool-install-prompt.js.map +1 -1
- package/dist/cli/weak-model-prompt.d.ts +16 -7
- package/dist/cli/weak-model-prompt.d.ts.map +1 -1
- package/dist/cli/weak-model-prompt.js +24 -8
- package/dist/cli/weak-model-prompt.js.map +1 -1
- package/dist/config/types.d.ts +44 -0
- package/dist/config/types.d.ts.map +1 -1
- package/dist/context/cache.d.ts +35 -1
- package/dist/context/cache.d.ts.map +1 -1
- package/dist/context/cache.js +31 -2
- package/dist/context/cache.js.map +1 -1
- package/dist/context/history.d.ts +21 -0
- package/dist/context/history.d.ts.map +1 -1
- package/dist/context/history.js +50 -0
- package/dist/context/history.js.map +1 -1
- package/dist/gateway/hook-contract.d.ts +173 -0
- package/dist/gateway/hook-contract.d.ts.map +1 -0
- package/dist/gateway/hook-contract.js +352 -0
- package/dist/gateway/hook-contract.js.map +1 -0
- package/dist/gateway/hooks.d.ts.map +1 -1
- package/dist/gateway/hooks.js +14 -0
- package/dist/gateway/hooks.js.map +1 -1
- package/dist/inference/interface.d.ts +17 -0
- package/dist/inference/interface.d.ts.map +1 -1
- package/dist/inference/model-catalog.d.ts +17 -0
- package/dist/inference/model-catalog.d.ts.map +1 -1
- package/dist/inference/model-catalog.js +42 -7
- package/dist/inference/model-catalog.js.map +1 -1
- package/dist/inference/model-id-validation.d.ts +57 -0
- package/dist/inference/model-id-validation.d.ts.map +1 -0
- package/dist/inference/model-id-validation.js +127 -0
- package/dist/inference/model-id-validation.js.map +1 -0
- package/dist/inference/openai-compat-adapter.d.ts.map +1 -1
- package/dist/inference/openai-compat-adapter.js +12 -1
- package/dist/inference/openai-compat-adapter.js.map +1 -1
- package/dist/inference/provider-catalog.d.ts +11 -0
- package/dist/inference/provider-catalog.d.ts.map +1 -1
- package/dist/inference/provider-catalog.js +80 -0
- package/dist/inference/provider-catalog.js.map +1 -1
- package/dist/inference/route-resolver.d.ts +2 -8
- package/dist/inference/route-resolver.d.ts.map +1 -1
- package/dist/inference/route-resolver.js +17 -0
- package/dist/inference/route-resolver.js.map +1 -1
- package/dist/inference/tool-call-utils.d.ts +4 -3
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +26 -12
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/inference/tools.d.ts +1 -0
- package/dist/inference/tools.d.ts.map +1 -1
- package/dist/inference/tools.js +20 -3
- package/dist/inference/tools.js.map +1 -1
- package/dist/learning/agentic-route-gate.d.ts +104 -0
- package/dist/learning/agentic-route-gate.d.ts.map +1 -0
- package/dist/learning/agentic-route-gate.js +125 -0
- package/dist/learning/agentic-route-gate.js.map +1 -0
- package/dist/learning/auto-router.d.ts +89 -0
- package/dist/learning/auto-router.d.ts.map +1 -1
- package/dist/learning/auto-router.js +188 -8
- package/dist/learning/auto-router.js.map +1 -1
- package/dist/learning/build-prerequisites.d.ts +68 -0
- package/dist/learning/build-prerequisites.d.ts.map +1 -0
- package/dist/learning/build-prerequisites.js +267 -0
- package/dist/learning/build-prerequisites.js.map +1 -0
- package/dist/learning/continuation-intent.d.ts +43 -0
- package/dist/learning/continuation-intent.d.ts.map +1 -0
- package/dist/learning/continuation-intent.js +82 -0
- package/dist/learning/continuation-intent.js.map +1 -0
- package/dist/learning/error-repair.d.ts +23 -1
- package/dist/learning/error-repair.d.ts.map +1 -1
- package/dist/learning/error-repair.js +58 -0
- package/dist/learning/error-repair.js.map +1 -1
- package/dist/learning/knowledge-base.d.ts +124 -0
- package/dist/learning/knowledge-base.d.ts.map +1 -0
- package/dist/learning/knowledge-base.js +337 -0
- package/dist/learning/knowledge-base.js.map +1 -0
- package/dist/learning/model-harness.d.ts +19 -0
- package/dist/learning/model-harness.d.ts.map +1 -1
- package/dist/learning/model-harness.js +27 -0
- package/dist/learning/model-harness.js.map +1 -1
- package/dist/learning/model-selection.d.ts +2 -1
- package/dist/learning/model-selection.d.ts.map +1 -1
- package/dist/learning/model-selection.js +35 -2
- package/dist/learning/model-selection.js.map +1 -1
- package/dist/learning/project-docs.d.ts +83 -0
- package/dist/learning/project-docs.d.ts.map +1 -0
- package/dist/learning/project-docs.js +132 -0
- package/dist/learning/project-docs.js.map +1 -0
- package/dist/learning/prompt-budget.d.ts +91 -0
- package/dist/learning/prompt-budget.d.ts.map +1 -0
- package/dist/learning/prompt-budget.js +99 -0
- package/dist/learning/prompt-budget.js.map +1 -0
- package/dist/learning/quota-ledger.d.ts +13 -0
- package/dist/learning/quota-ledger.d.ts.map +1 -1
- package/dist/learning/quota-ledger.js +28 -0
- package/dist/learning/quota-ledger.js.map +1 -1
- package/dist/learning/reasoning-trace.d.ts +55 -2
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +48 -0
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +16 -3
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/routing-history.d.ts +13 -0
- package/dist/learning/routing-history.d.ts.map +1 -1
- package/dist/learning/routing-history.js.map +1 -1
- package/dist/learning/seeded-benchmark.d.ts +14 -0
- package/dist/learning/seeded-benchmark.d.ts.map +1 -1
- package/dist/learning/seeded-benchmark.js +43 -4
- package/dist/learning/seeded-benchmark.js.map +1 -1
- package/dist/learning/turn-report.d.ts +78 -0
- package/dist/learning/turn-report.d.ts.map +1 -0
- package/dist/learning/turn-report.js +140 -0
- package/dist/learning/turn-report.js.map +1 -0
- package/dist/security/secret-scan.d.ts +156 -0
- package/dist/security/secret-scan.d.ts.map +1 -0
- package/dist/security/secret-scan.js +525 -0
- package/dist/security/secret-scan.js.map +1 -0
- package/dist/tools/edit-verification.d.ts +2 -11
- package/dist/tools/edit-verification.d.ts.map +1 -1
- package/dist/tools/edit-verification.js +33 -1
- package/dist/tools/edit-verification.js.map +1 -1
- package/dist/tools/loop-skill-hint.d.ts +51 -0
- package/dist/tools/loop-skill-hint.d.ts.map +1 -1
- package/dist/tools/loop-skill-hint.js +87 -2
- package/dist/tools/loop-skill-hint.js.map +1 -1
- package/dist/tools/plan-store.d.ts +126 -3
- package/dist/tools/plan-store.d.ts.map +1 -1
- package/dist/tools/plan-store.js +290 -6
- package/dist/tools/plan-store.js.map +1 -1
- package/dist/tools/registry.d.ts +20 -1
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +165 -4
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/remediation-ladder.d.ts +91 -0
- package/dist/tools/remediation-ladder.d.ts.map +1 -0
- package/dist/tools/remediation-ladder.js +308 -0
- package/dist/tools/remediation-ladder.js.map +1 -0
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +60 -2
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/skills-hub.d.ts +7 -0
- package/dist/tools/skills-hub.d.ts.map +1 -1
- package/dist/tools/skills-hub.js +11 -2
- package/dist/tools/skills-hub.js.map +1 -1
- package/dist/tools/step-artifact.d.ts +35 -0
- package/dist/tools/step-artifact.d.ts.map +1 -0
- package/dist/tools/step-artifact.js +154 -0
- package/dist/tools/step-artifact.js.map +1 -0
- package/dist/tools/tool-loop.d.ts +35 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +284 -9
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/tools/toolsets.js +4 -4
- package/dist/tools/toolsets.js.map +1 -1
- package/dist/tools/worktree.d.ts +11 -2
- package/dist/tools/worktree.d.ts.map +1 -1
- package/dist/tools/worktree.js +11 -2
- package/dist/tools/worktree.js.map +1 -1
- package/dist/utils/effect-verification.js +1 -1
- package/dist/utils/effect-verification.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +21 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +33 -5
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/server.d.ts +29 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +410 -5
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +137 -0
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.d.ts +7 -0
- package/dist/web-dashboard/workspace-guard.d.ts.map +1 -1
- package/dist/web-dashboard/workspace-guard.js +70 -0
- package/dist/web-dashboard/workspace-guard.js.map +1 -1
- package/package.json +4 -1
- package/src/web-dashboard/public/assets/index-DhiuR4S2.js +210 -0
- package/src/web-dashboard/public/assets/index-DhiuR4S2.js.map +1 -0
- package/src/web-dashboard/public/assets/index-tK8Y2qHB.css +1 -0
- package/src/web-dashboard/public/index.html +2 -2
- package/src/web-dashboard/public/assets/index-BdKf5Xw2.js +0 -207
- package/src/web-dashboard/public/assets/index-BdKf5Xw2.js.map +0 -1
- package/src/web-dashboard/public/assets/index-C9eBskc9.css +0 -1
package/dist/tools/tool-loop.js
CHANGED
|
@@ -49,6 +49,8 @@ import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, veri
|
|
|
49
49
|
import { effectiveToolJsonSchemas, coreToolJsonSchemas, isToolEnabled, toolsetForTool } from './toolsets.js';
|
|
50
50
|
import { deliverablesNamedIn, recordStepHandoff } from '../agents/step-handoff.js';
|
|
51
51
|
import { fenceUntrustedToolOutput } from './untrusted-content.js';
|
|
52
|
+
import { installableToolsFromFailure, resolveInstallCommand, toolTakeoverInstruction, } from '../cli/tool-install-prompt.js';
|
|
53
|
+
import { matchPrerequisiteSignatures, prerequisiteTakeoverInstruction, checkProjectPrerequisites, createNodePrereqFs, formatPreflightFindings, } from '../learning/build-prerequisites.js';
|
|
52
54
|
/**
|
|
53
55
|
* Tools that MUTATE the workspace. A refusal of one of these is the only
|
|
54
56
|
* refusal that leaves work undone — a declined `read_file` costs a step, a
|
|
@@ -61,6 +63,19 @@ const MUTATING_TOOL_NAMES = new Set([
|
|
|
61
63
|
'run_terminal',
|
|
62
64
|
'run_cli',
|
|
63
65
|
]);
|
|
66
|
+
/**
|
|
67
|
+
* The subset of mutating tools the PLAN gate HARD-BLOCKS on the first call —
|
|
68
|
+
* the ones that change the workspace's FILES. `run_terminal` / `run_cli` are
|
|
69
|
+
* deliberately excluded from the BLOCK (they still trigger the nudge): a
|
|
70
|
+
* terminal command is as often a test, a build or an inspection as it is an
|
|
71
|
+
* edit, and refusing it outright would stop the very step a plan is meant to
|
|
72
|
+
* reach. The nudge still tells the model to plan first.
|
|
73
|
+
*/
|
|
74
|
+
const PLAN_GATED_TOOL_NAMES = new Set([
|
|
75
|
+
'write_file',
|
|
76
|
+
'edit_file',
|
|
77
|
+
'propose_change',
|
|
78
|
+
]);
|
|
64
79
|
/** The path a mutating call was aimed at, when it named one. */
|
|
65
80
|
function mutatedPathOf(args) {
|
|
66
81
|
const a = args;
|
|
@@ -438,6 +453,15 @@ export const NO_PROGRESS_STALL_STEPS = 4;
|
|
|
438
453
|
* where an extra pass earns its latency.
|
|
439
454
|
*/
|
|
440
455
|
export const SELF_REVIEW_MIN_STEPS = 6;
|
|
456
|
+
/**
|
|
457
|
+
* S5 — how many `plan_todo` UPDATE calls one turn may make. The guard refuses
|
|
458
|
+
* repeated CREATES (the observed planner loop) but must allow updates, because
|
|
459
|
+
* an update is how the user's checklist advances; this cap only stops a model
|
|
460
|
+
* that replaces doing the work with spamming status changes. It sits well above
|
|
461
|
+
* a genuine multi-step job (the largest real plans in the tree are single
|
|
462
|
+
* digits) and far below a loop.
|
|
463
|
+
*/
|
|
464
|
+
export const PLAN_TODO_UPDATE_CAP = 64;
|
|
441
465
|
/** Default continuations granted per turn when the option is omitted. */
|
|
442
466
|
export const DEFAULT_MAX_CONTINUATIONS = 2;
|
|
443
467
|
/** Default extra steps granted per continuation. */
|
|
@@ -537,6 +561,12 @@ async function runToolLoopInner(opts, progress) {
|
|
|
537
561
|
const maxParallelReads = Math.max(1, opts.maxParallelReads ?? MAX_PARALLEL_READS);
|
|
538
562
|
const followups = [];
|
|
539
563
|
const toolCallsRun = [];
|
|
564
|
+
// S5 — the plan_todo loop guard is split by ACTION (see the guard below):
|
|
565
|
+
// repeated CREATES were the observed planner loop; UPDATES are the tracking
|
|
566
|
+
// that keeps the user's checklist honest. Counted across the whole turn, like
|
|
567
|
+
// `toolCallsRun`, because a multi-step job runs many steps in one turn.
|
|
568
|
+
let planTodoCreates = 0;
|
|
569
|
+
let planTodoUpdates = 0;
|
|
540
570
|
// Every collected suggestion passes through the shared normalizer, so the
|
|
541
571
|
// loop's output is ALWAYS clean + structured (1–3 items, deduped, no leaked
|
|
542
572
|
// tool JSON, capped prompt/label) regardless of what the model emitted —
|
|
@@ -715,6 +745,18 @@ async function runToolLoopInner(opts, progress) {
|
|
|
715
745
|
// with an empty/bounded response (the "where is the essay?" bug).
|
|
716
746
|
let lastContent = '';
|
|
717
747
|
let bounded = false;
|
|
748
|
+
// D2.2 — has this turn already pre-flighted the project's build prerequisites?
|
|
749
|
+
// Checked once, before the FIRST build command the turn plans to run.
|
|
750
|
+
let prerequisitesPreflighted = false;
|
|
751
|
+
// E2 — has this turn already spent its one bounded "declare a plan first"
|
|
752
|
+
// nudge? Bounded once, like every other gate.
|
|
753
|
+
let planNudged = false;
|
|
754
|
+
// E2 (hard) — has this turn already spent its one bounded BLOCK of the first
|
|
755
|
+
// mutation? The plan gate is a hard requirement the first time it fires and a
|
|
756
|
+
// pure nudge after that: the first mutating batch is refused (see the gate
|
|
757
|
+
// below), and the second attempt runs even without a plan — a bounded block,
|
|
758
|
+
// never a wall.
|
|
759
|
+
let planBlocked = false;
|
|
718
760
|
/**
|
|
719
761
|
* Should the SELF-REVIEW gate fire now? Returns the correction to inject, or
|
|
720
762
|
* null. Extracted so the two end-of-turn exits ask the question identically —
|
|
@@ -937,7 +979,12 @@ async function runToolLoopInner(opts, progress) {
|
|
|
937
979
|
// isThinkOnlyResponse: continue instead of ending).
|
|
938
980
|
if (deps.isThinkOnly ? deps.isThinkOnly(response.content) : isThinkOnlyResponse(response.content)) {
|
|
939
981
|
// Feed an empty assistant step so the model continues in-context.
|
|
940
|
-
thread.push({
|
|
982
|
+
thread.push({
|
|
983
|
+
role: 'assistant',
|
|
984
|
+
content: response.content,
|
|
985
|
+
// Keep the thinking model's reasoning with the turn it belongs to.
|
|
986
|
+
...(response.reasoningContent ? { reasoningContent: response.reasoningContent } : {}),
|
|
987
|
+
});
|
|
941
988
|
thinkContinues += 1;
|
|
942
989
|
// AN EMPTY RESPONSE IS A FAILURE, NOT REASONING. `isThinkOnlyResponse('')`
|
|
943
990
|
// returns true, so a provider returning nothing (the loop arm falls over
|
|
@@ -1224,20 +1271,67 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1224
1271
|
arguments: JSON.stringify(tc.arguments),
|
|
1225
1272
|
...(tc.providerMeta ? { providerMeta: tc.providerMeta } : {}),
|
|
1226
1273
|
})),
|
|
1274
|
+
// Thinking-mode reasoning is echoed back with the assistant turn that
|
|
1275
|
+
// carried these calls — DeepSeek v4 rejects the next request without it.
|
|
1276
|
+
...(response.reasoningContent ? { reasoningContent: response.reasoningContent } : {}),
|
|
1227
1277
|
});
|
|
1228
1278
|
let endedAfterConcluding = false;
|
|
1279
|
+
// ── Provider-protocol safety: nudges decided while PLANNING must queue ──
|
|
1280
|
+
// An assistant message that carries `tool_calls` MUST be followed
|
|
1281
|
+
// immediately by a `tool` result for EVERY call id, before any other role.
|
|
1282
|
+
// A strict OpenAI-compatible API (DeepSeek native) rejects the NEXT request
|
|
1283
|
+
// otherwise: "An assistant message with 'tool_calls' must be followed by
|
|
1284
|
+
// tool messages responding to each 'tool_call_id'." The pre-flight and plan
|
|
1285
|
+
// gates decide during planning — before the results exist — so they push
|
|
1286
|
+
// their nudge into this queue and it is flushed after the results, keeping
|
|
1287
|
+
// the assistant→tool group intact. (Without this, a pinned/strict run died
|
|
1288
|
+
// on the third model call and looked like a provider fault.)
|
|
1289
|
+
const deferredUserNudges = [];
|
|
1229
1290
|
const plans = toolCalls.map((call) => {
|
|
1230
1291
|
const tool = getTool(call.name);
|
|
1231
1292
|
const priorSameTool = toolCallsRun.filter((t) => t === call.name).length;
|
|
1232
1293
|
toolCallsRun.push(call.name);
|
|
1233
|
-
// S5 — PLANNER LOOP GUARD
|
|
1234
|
-
//
|
|
1235
|
-
//
|
|
1236
|
-
|
|
1237
|
-
|
|
1238
|
-
|
|
1239
|
-
|
|
1240
|
-
|
|
1294
|
+
// S5 — PLANNER LOOP GUARD, now split by ACTION.
|
|
1295
|
+
//
|
|
1296
|
+
// The guard exists because of a real failure: the model called plan_todo
|
|
1297
|
+
// six times in one turn (trace-1788059239352-k7zl03 — 15.6K tokens,
|
|
1298
|
+
// 2m25s, FAILED) instead of doing the work. But `create` and `update` are
|
|
1299
|
+
// the SAME tool, and the original `priorSameTool >= 1` check refused BOTH
|
|
1300
|
+
// — so a plan could be declared once and then never advanced, and the
|
|
1301
|
+
// checklist the user watches went stale the moment the first step closed.
|
|
1302
|
+
// Only repeated CREATES are the loop; updates ARE the tracking. Anything
|
|
1303
|
+
// that is not an explicit update (absent/malformed action) is counted as
|
|
1304
|
+
// a create, so the original protection is unchanged for the loop case.
|
|
1305
|
+
if (call.name === 'plan_todo') {
|
|
1306
|
+
const action = (() => {
|
|
1307
|
+
try {
|
|
1308
|
+
const raw = call.arguments;
|
|
1309
|
+
const obj = typeof raw === 'string' ? JSON.parse(raw) : raw;
|
|
1310
|
+
return obj && typeof obj === 'object'
|
|
1311
|
+
? String(obj.action ?? '')
|
|
1312
|
+
: undefined;
|
|
1313
|
+
}
|
|
1314
|
+
catch {
|
|
1315
|
+
return undefined;
|
|
1316
|
+
}
|
|
1317
|
+
})();
|
|
1318
|
+
const isUpdate = action === 'update';
|
|
1319
|
+
if (!isUpdate && planTodoCreates >= 1) {
|
|
1320
|
+
return {
|
|
1321
|
+
call,
|
|
1322
|
+
refuse: 'Error: a plan already exists for this turn — do NOT declare it again. Advance it instead: call plan_todo with action "update", the step id and its status, then keep doing the work.',
|
|
1323
|
+
};
|
|
1324
|
+
}
|
|
1325
|
+
if (isUpdate && planTodoUpdates >= PLAN_TODO_UPDATE_CAP) {
|
|
1326
|
+
return {
|
|
1327
|
+
call,
|
|
1328
|
+
refuse: `Error: plan_todo has been updated ${PLAN_TODO_UPDATE_CAP} times this turn — stop updating the plan and finish the remaining work.`,
|
|
1329
|
+
};
|
|
1330
|
+
}
|
|
1331
|
+
if (isUpdate)
|
|
1332
|
+
planTodoUpdates += 1;
|
|
1333
|
+
else
|
|
1334
|
+
planTodoCreates += 1;
|
|
1241
1335
|
}
|
|
1242
1336
|
if (tool?.category === 'pipeline' && priorSameTool >= 1) {
|
|
1243
1337
|
// The guard this replaces tested the literal name `pipeline`, which is
|
|
@@ -1431,6 +1525,118 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1431
1525
|
toolSpan?.end({ ok: false, message: 'tool outcome was never reported' });
|
|
1432
1526
|
}
|
|
1433
1527
|
};
|
|
1528
|
+
// ── PROJECT PREREQUISITE PRE-FLIGHT (D2.2) ──────────────────────────
|
|
1529
|
+
// Before the FIRST build of the turn, check the project's markers so a
|
|
1530
|
+
// missing prerequisite (build.rs, a Cargo feature, an icon) is handed over
|
|
1531
|
+
// BEFORE a failed build — not after ten identical retries. Best-effort and
|
|
1532
|
+
// bounded: once per turn, only when a build is actually planned.
|
|
1533
|
+
if (!prerequisitesPreflighted) {
|
|
1534
|
+
const buildPlan = plans.find((p) => {
|
|
1535
|
+
if (p.refuse !== undefined)
|
|
1536
|
+
return false;
|
|
1537
|
+
if (p.call.name !== 'run_terminal' && p.call.name !== 'terminal')
|
|
1538
|
+
return false;
|
|
1539
|
+
try {
|
|
1540
|
+
const raw = p.call.arguments;
|
|
1541
|
+
const obj = typeof raw === 'string' ? JSON.parse(raw) : raw;
|
|
1542
|
+
const cmd = String(obj?.command ?? '');
|
|
1543
|
+
return isBuildCommand(cmd);
|
|
1544
|
+
}
|
|
1545
|
+
catch {
|
|
1546
|
+
return false;
|
|
1547
|
+
}
|
|
1548
|
+
});
|
|
1549
|
+
if (buildPlan) {
|
|
1550
|
+
prerequisitesPreflighted = true;
|
|
1551
|
+
try {
|
|
1552
|
+
const findings = checkProjectPrerequisites(createNodePrereqFs(opts.context?.cwd || process.cwd()));
|
|
1553
|
+
if (findings.length > 0) {
|
|
1554
|
+
deferredUserNudges.push(formatPreflightFindings(findings));
|
|
1555
|
+
deps.onEvent?.(` 🧱 pre-flight: ${findings.length} missing project prerequisite(s) — handed the exact fix before building.`);
|
|
1556
|
+
traceEvent({
|
|
1557
|
+
kind: 'gate',
|
|
1558
|
+
gate: 'prerequisite',
|
|
1559
|
+
summary: `pre-flight found ${findings.length} missing project prerequisite(s) before the build: ` +
|
|
1560
|
+
findings.map((f) => f.id).join(', '),
|
|
1561
|
+
});
|
|
1562
|
+
}
|
|
1563
|
+
}
|
|
1564
|
+
catch {
|
|
1565
|
+
// A pre-flight must never break the turn.
|
|
1566
|
+
}
|
|
1567
|
+
}
|
|
1568
|
+
}
|
|
1569
|
+
// ── PLAN GATE (E2) ──────────────────────────────────────────────────
|
|
1570
|
+
// A workspace-directing turn about to MUTATE without having declared a plan
|
|
1571
|
+
// is stopped ONCE: its first FILE mutation is refused and the model is told
|
|
1572
|
+
// to plan first (`plan_todo`). Terminal commands still trigger the nudge but
|
|
1573
|
+
// are not blocked (see PLAN_GATED_TOOL_NAMES) — a build or a test is a step
|
|
1574
|
+
// a plan is meant to reach, not a change to gate. Bounded by design: the
|
|
1575
|
+
// SECOND attempt runs even without a plan (see `planBlocked`), so this guides
|
|
1576
|
+
// the model into the plan → track → verify contract instead of walling the
|
|
1577
|
+
// turn off. A batch that DECLARES a plan itself is never blocked (the model
|
|
1578
|
+
// is already doing the right thing), and `requirePlan:false` leaves the gate
|
|
1579
|
+
// byte-identical to not existing.
|
|
1580
|
+
const declaresPlan = !opts.context?.planStore?.snapshot?.() && plans.some((p) => {
|
|
1581
|
+
if (p.call.name !== 'plan_todo')
|
|
1582
|
+
return false;
|
|
1583
|
+
try {
|
|
1584
|
+
const raw = p.call.arguments;
|
|
1585
|
+
const obj = typeof raw === 'string' ? JSON.parse(raw) : raw;
|
|
1586
|
+
const action = obj && typeof obj === 'object' ? String(obj.action ?? '') : '';
|
|
1587
|
+
return action !== 'update';
|
|
1588
|
+
}
|
|
1589
|
+
catch {
|
|
1590
|
+
// Malformed args still count as a create attempt — the plan_todo guard
|
|
1591
|
+
// above already handles a repeated create, and a first one is honored.
|
|
1592
|
+
return true;
|
|
1593
|
+
}
|
|
1594
|
+
});
|
|
1595
|
+
if (!planNudged &&
|
|
1596
|
+
!declaresPlan &&
|
|
1597
|
+
opts.requirePlan !== false &&
|
|
1598
|
+
schemas.length > 0 &&
|
|
1599
|
+
!opts.context?.planStore?.snapshot?.() &&
|
|
1600
|
+
plans.some((p) => p.refuse === undefined && MUTATING_TOOL_NAMES.has(p.call.name)) &&
|
|
1601
|
+
requestRequiresWorkspaceAction(requestText, authorization.authorized)) {
|
|
1602
|
+
planNudged = true;
|
|
1603
|
+
// Hard requirement, spent ONCE: refuse the first batch's FILE mutations
|
|
1604
|
+
// so nothing is written before a plan exists. It bumps the step budget by
|
|
1605
|
+
// one because it forces a genuine re-answer (the model must declare the
|
|
1606
|
+
// plan and retry), exactly like the zero-action nudge.
|
|
1607
|
+
const blocked = [];
|
|
1608
|
+
if (!planBlocked) {
|
|
1609
|
+
planBlocked = true;
|
|
1610
|
+
for (const p of plans) {
|
|
1611
|
+
if (p.refuse === undefined && PLAN_GATED_TOOL_NAMES.has(p.call.name)) {
|
|
1612
|
+
p.refuse =
|
|
1613
|
+
'Error: declare a plan first via plan_todo — call it with 2–6 ordered steps (action "create"), then retry this change. The plan is how the turn is tracked and verified.';
|
|
1614
|
+
blocked.push(p.call.name);
|
|
1615
|
+
}
|
|
1616
|
+
}
|
|
1617
|
+
if (blocked.length > 0)
|
|
1618
|
+
stepLimit += 1;
|
|
1619
|
+
for (const name of blocked) {
|
|
1620
|
+
traceEvent({
|
|
1621
|
+
kind: 'refusal',
|
|
1622
|
+
gate: 'plan',
|
|
1623
|
+
tool: name,
|
|
1624
|
+
summary: `refused ${name}: no plan declared — the first mutation is blocked once so the model plans before it changes files`,
|
|
1625
|
+
});
|
|
1626
|
+
}
|
|
1627
|
+
}
|
|
1628
|
+
deps.onEvent?.(blocked.length > 0
|
|
1629
|
+
? ' 🗺️ No plan declared — blocking the first change until the model plans.'
|
|
1630
|
+
: ' 🗺️ No plan declared — asking the model to plan before it changes files.');
|
|
1631
|
+
traceEvent({
|
|
1632
|
+
kind: 'gate',
|
|
1633
|
+
gate: 'plan',
|
|
1634
|
+
summary: blocked.length > 0
|
|
1635
|
+
? `a workspace-directing turn was about to mutate without a plan — the first mutation batch (${blocked.join(', ')}) was refused once and the model asked to declare a plan first`
|
|
1636
|
+
: 'a workspace-directing turn was about to mutate without a plan — one bounded nudge to declare one first',
|
|
1637
|
+
});
|
|
1638
|
+
deferredUserNudges.push(planRequiredNudge(currentAsk(opts)));
|
|
1639
|
+
}
|
|
1434
1640
|
const executed = new Array(plans.length).fill('');
|
|
1435
1641
|
for (let i = 0; i < plans.length;) {
|
|
1436
1642
|
// A run of consecutive read-only calls is ONE bounded fan-out; any
|
|
@@ -1581,6 +1787,54 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1581
1787
|
const parallelTip = parallel.note(call.name, !resultText.startsWith('Error:'));
|
|
1582
1788
|
if (parallelTip)
|
|
1583
1789
|
resultText = `${resultText}\n\n${parallelTip}`;
|
|
1790
|
+
// ── MISSING-PREREQUISITE TAKEOVER ──────────────────────────────────
|
|
1791
|
+
// A `command not found` (exit 127) for a KNOWN, installable system tool
|
|
1792
|
+
// is a STEP, not a wall — but a model can read it as an environment
|
|
1793
|
+
// limit and stop: "I cannot install system-level software on your host
|
|
1794
|
+
// machine — I am physically unable" (live: trace-1791127992452-qzgodi,
|
|
1795
|
+
// where a user who had explicitly granted terminal permission was handed
|
|
1796
|
+
// a manual `curl … | sh` step instead of the install being done). Inject
|
|
1797
|
+
// the exact install command and forbid the refusal, deterministically, so
|
|
1798
|
+
// the NEXT model step acts. Advisory text only — the loop never runs the
|
|
1799
|
+
// install itself; that stays the model's governed `run_terminal` call.
|
|
1800
|
+
if (!ranOk && (call.name === 'run_terminal' || call.name === 'terminal')) {
|
|
1801
|
+
const failedCommand = String(call.arguments?.command ?? '');
|
|
1802
|
+
// EVERY installable tool the failing line needs, not just the first —
|
|
1803
|
+
// a `cd app && cargo build && cmake .` turn should hand the model all
|
|
1804
|
+
// the prerequisites in one takeover instead of one per round-trip.
|
|
1805
|
+
const missing = installableToolsFromFailure(failedCommand, rawResult);
|
|
1806
|
+
if (missing.length > 0) {
|
|
1807
|
+
const installCommands = missing.map((t) => resolveInstallCommand(t));
|
|
1808
|
+
resultText = `${resultText}\n\n${toolTakeoverInstruction(missing, installCommands)}`;
|
|
1809
|
+
deps.onEvent?.(` 🛠️ ${missing.map((t) => `'${t}'`).join(', ')} missing and installable — handing the model the install command(s) to run.`);
|
|
1810
|
+
traceEvent({
|
|
1811
|
+
kind: 'gate',
|
|
1812
|
+
gate: 'tool-takeover',
|
|
1813
|
+
summary: `${missing.join(', ')} missing (command not found) and installable — the model is ` +
|
|
1814
|
+
'told to install them itself instead of declaring itself unable',
|
|
1815
|
+
});
|
|
1816
|
+
}
|
|
1817
|
+
// ── PROJECT PREREQUISITE BACKSTOP (D2.3) ──────────────────────────
|
|
1818
|
+
// No missing BINARY, but this failure may be a missing project FILE/
|
|
1819
|
+
// FEATURE (the other half of the same wall): `src-tauri/build.rs`, a
|
|
1820
|
+
// Cargo `custom-protocol` feature, an icon — none of which is a tool to
|
|
1821
|
+
// install, so the takeover above never fired and the live macOS turn
|
|
1822
|
+
// re-ran `npx tauri build` ~10 times. Match the OUTPUT SIGNATURE, hand
|
|
1823
|
+
// the model the exact fix on the FIRST failure, and record it.
|
|
1824
|
+
if (missing.length === 0) {
|
|
1825
|
+
const rules = matchPrerequisiteSignatures(rawResult);
|
|
1826
|
+
if (rules.length > 0) {
|
|
1827
|
+
resultText = `${resultText}\n\n${prerequisiteTakeoverInstruction(rules)}`;
|
|
1828
|
+
deps.onEvent?.(` 🧱 build prerequisite signature matched (${rules.map((r) => r.id).join(', ')}) — handing the model the exact fix.`);
|
|
1829
|
+
traceEvent({
|
|
1830
|
+
kind: 'gate',
|
|
1831
|
+
gate: 'prerequisite',
|
|
1832
|
+
summary: `build failed on a missing PROJECT prerequisite (${rules.map((r) => r.id).join(', ')}) — ` +
|
|
1833
|
+
'the model is given the exact fix instead of retrying the identical command',
|
|
1834
|
+
});
|
|
1835
|
+
}
|
|
1836
|
+
}
|
|
1837
|
+
}
|
|
1584
1838
|
delivered[i] = resultText;
|
|
1585
1839
|
// G18 — record what this call DID, in the model's own call order, with the
|
|
1586
1840
|
// evidence a later reader needs: the args the gate saw, the first line of
|
|
@@ -1602,6 +1856,11 @@ async function runToolLoopInner(opts, progress) {
|
|
|
1602
1856
|
});
|
|
1603
1857
|
thread.push({ role: 'tool', toolCallId: call.id, content: resultText });
|
|
1604
1858
|
}
|
|
1859
|
+
// Flush the planning-time nudges now that the assistant's tool_calls group is
|
|
1860
|
+
// complete (every call id has a `tool` result). This is the first point a
|
|
1861
|
+
// user/system message may safely follow the batch.
|
|
1862
|
+
for (const nudge of deferredUserNudges)
|
|
1863
|
+
thread.push({ role: 'user', content: nudge });
|
|
1605
1864
|
// Track the generalized stall: a step that RAN tools but succeeded at none
|
|
1606
1865
|
// of them extends the streak; any success clears it. A text-only step (no
|
|
1607
1866
|
// tools) leaves it as it was — the model is thinking, not failing.
|
|
@@ -2298,6 +2557,22 @@ function zeroActionGateApplies(opts, progress, schemaCount, requestText, authori
|
|
|
2298
2557
|
return false;
|
|
2299
2558
|
return requestRequiresWorkspaceAction(requestText, authorized);
|
|
2300
2559
|
}
|
|
2560
|
+
/**
|
|
2561
|
+
* The bounded PLAN-REQUIRED nudge (E2) — one advisory step telling the model to
|
|
2562
|
+
* declare a short plan before it mutates.
|
|
2563
|
+
*
|
|
2564
|
+
* Names the ask, states the contract (plan first, then work), and closes the
|
|
2565
|
+
* failure mode the nudge exists for: planning that REPLACES the work; the last
|
|
2566
|
+
* line forbids exactly that dormancy.
|
|
2567
|
+
*/
|
|
2568
|
+
export function planRequiredNudge(ask) {
|
|
2569
|
+
const quoted = (ask || '').trim().replace(/\s+/g, ' ').slice(0, 300);
|
|
2570
|
+
return ('Before you change anything, declare a short PLAN for this turn — call `plan_todo` with 2–6 ordered steps' +
|
|
2571
|
+
' (create the plan once, then advance it with action "update" as you go).' +
|
|
2572
|
+
(quoted ? ` The request: "${quoted}".` : '') +
|
|
2573
|
+
'\nA plan is how the user can follow along and how the turn is verified — keep it accurate, do not pad it,' +
|
|
2574
|
+
' and do NOT let planning replace the work: declare it, then do the first step now.');
|
|
2575
|
+
}
|
|
2301
2576
|
/**
|
|
2302
2577
|
* The bounded ZERO-ACTION correction — the loop telling the model, in one step,
|
|
2303
2578
|
* that a directed request has not been touched yet.
|