agent-nuvira 3.2.0 → 3.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/code-assessment/SKILL.md +10 -1
- package/.agents/skills/index.json +1 -1
- package/README.md +42 -1
- package/dist/agents/agents/writer.d.ts.map +1 -1
- package/dist/agents/agents/writer.js +7 -1
- package/dist/agents/agents/writer.js.map +1 -1
- package/dist/cli/chat.d.ts +6 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +45 -1
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts.map +1 -1
- package/dist/cli/cli-program.js +6 -2
- package/dist/cli/cli-program.js.map +1 -1
- package/dist/cli/commands.d.ts +2 -1
- package/dist/cli/commands.d.ts.map +1 -1
- package/dist/cli/commands.js +3 -2
- package/dist/cli/commands.js.map +1 -1
- package/dist/cli/config.js +2 -2
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/dashboard.d.ts +26 -0
- package/dist/cli/dashboard.d.ts.map +1 -1
- package/dist/cli/dashboard.js +123 -3
- package/dist/cli/dashboard.js.map +1 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +48 -3
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/intent.d.ts +1 -1
- package/dist/cli/intent.js +2 -2
- package/dist/cli/intent.js.map +1 -1
- package/dist/cli/loop-executor.d.ts +24 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +161 -1
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/process-control.d.ts +15 -0
- package/dist/cli/process-control.d.ts.map +1 -1
- package/dist/cli/process-control.js +38 -14
- package/dist/cli/process-control.js.map +1 -1
- package/dist/cli/trace.d.ts +19 -0
- package/dist/cli/trace.d.ts.map +1 -1
- package/dist/cli/trace.js +82 -2
- package/dist/cli/trace.js.map +1 -1
- package/dist/forwarded.d.ts +2 -0
- package/dist/forwarded.d.ts.map +1 -0
- package/dist/forwarded.js +2 -0
- package/dist/forwarded.js.map +1 -0
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +22 -3
- package/dist/gateway/registry.js.map +1 -1
- package/dist/inference/tool-call-utils.d.ts +54 -0
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +216 -4
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/learning/autonomy-policy.d.ts +42 -0
- package/dist/learning/autonomy-policy.d.ts.map +1 -1
- package/dist/learning/autonomy-policy.js +115 -19
- package/dist/learning/autonomy-policy.js.map +1 -1
- package/dist/learning/cost-tracker.d.ts +14 -0
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +22 -0
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/deliverable-class.d.ts +48 -0
- package/dist/learning/deliverable-class.d.ts.map +1 -1
- package/dist/learning/deliverable-class.js +72 -0
- package/dist/learning/deliverable-class.js.map +1 -1
- package/dist/learning/engine-router.d.ts +12 -1
- package/dist/learning/engine-router.d.ts.map +1 -1
- package/dist/learning/engine-router.js +28 -0
- package/dist/learning/engine-router.js.map +1 -1
- package/dist/learning/eval-framework.d.ts +24 -0
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +291 -0
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/intent-envelope.d.ts +162 -0
- package/dist/learning/intent-envelope.d.ts.map +1 -0
- package/dist/learning/intent-envelope.js +235 -0
- package/dist/learning/intent-envelope.js.map +1 -0
- package/dist/learning/long-form.d.ts.map +1 -1
- package/dist/learning/long-form.js +12 -1
- package/dist/learning/long-form.js.map +1 -1
- package/dist/learning/reasoning-trace.d.ts +89 -0
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +49 -0
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/run-trace.d.ts +164 -0
- package/dist/learning/run-trace.d.ts.map +1 -0
- package/dist/learning/run-trace.js +342 -0
- package/dist/learning/run-trace.js.map +1 -0
- package/dist/learning/skill-store.d.ts.map +1 -1
- package/dist/learning/skill-store.js +5 -1
- package/dist/learning/skill-store.js.map +1 -1
- package/dist/learning/skill-types.d.ts +14 -0
- package/dist/learning/skill-types.d.ts.map +1 -1
- package/dist/learning/skill-types.js +19 -0
- package/dist/learning/skill-types.js.map +1 -1
- package/dist/learning/unattended-job.d.ts +57 -0
- package/dist/learning/unattended-job.d.ts.map +1 -1
- package/dist/learning/unattended-job.js +92 -0
- package/dist/learning/unattended-job.js.map +1 -1
- package/dist/nlu/conversation-gate.d.ts +9 -0
- package/dist/nlu/conversation-gate.d.ts.map +1 -1
- package/dist/nlu/conversation-gate.js +38 -17
- package/dist/nlu/conversation-gate.js.map +1 -1
- package/dist/resources/command-manifest.json +10 -10
- package/dist/skills/bundled-skills.d.ts.map +1 -1
- package/dist/skills/bundled-skills.js +13 -2
- package/dist/skills/bundled-skills.js.map +1 -1
- package/dist/tools/coding-tools.d.ts.map +1 -1
- package/dist/tools/coding-tools.js +33 -8
- package/dist/tools/coding-tools.js.map +1 -1
- package/dist/tools/edit-verification.d.ts +95 -0
- package/dist/tools/edit-verification.d.ts.map +1 -1
- package/dist/tools/edit-verification.js +238 -16
- package/dist/tools/edit-verification.js.map +1 -1
- package/dist/tools/registry.d.ts +31 -2
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +44 -5
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/run-cli.d.ts.map +1 -1
- package/dist/tools/run-cli.js +15 -2
- package/dist/tools/run-cli.js.map +1 -1
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +23 -4
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/skill-tool.d.ts.map +1 -1
- package/dist/tools/skill-tool.js +5 -0
- package/dist/tools/skill-tool.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +102 -1
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +480 -13
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +7 -0
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +41 -3
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/server.d.ts +46 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +115 -1
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/admin-auth.d.ts +49 -2
- package/dist/web-dashboard/src/admin-auth.d.ts.map +1 -1
- package/dist/web-dashboard/src/admin-auth.js +64 -3
- package/dist/web-dashboard/src/admin-auth.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +41 -0
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/package.json +5 -2
- package/src/web-dashboard/public/assets/index-CjvoBhlz.js +207 -0
- package/src/web-dashboard/public/assets/index-CjvoBhlz.js.map +1 -0
- package/src/web-dashboard/public/index.html +1 -1
- package/src/web-dashboard/public/assets/index-kCUkORm7.js +0 -207
- package/src/web-dashboard/public/assets/index-kCUkORm7.js.map +0 -1
package/dist/tools/tool-loop.js
CHANGED
|
@@ -16,13 +16,16 @@
|
|
|
16
16
|
* blocks after its response text (contract in TOOL_CONTRACT_JSON).
|
|
17
17
|
*/
|
|
18
18
|
import { getTool, toolJsonSchemas } from './registry.js';
|
|
19
|
-
import { detectPermissionSeeking, requestAuthorizesWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
|
|
19
|
+
import { detectPermissionSeeking, isAffirmativeReply, replyAsksTheReader, requestAuthorizesWrites, stripTrailingPermissionSeek, } from '../learning/autonomy-policy.js';
|
|
20
|
+
import { envelopeFromPlan, envelopeFromRequest, getEnvelope, grantEnvelope, isEnvelopeKey, } from '../learning/intent-envelope.js';
|
|
21
|
+
import { detectProcessComplaint, isTraceKey, repeatNudge, runTraceFor, RunTrace, } from '../learning/run-trace.js';
|
|
22
|
+
import { wantsAuthoredArtifact } from '../learning/deliverable-class.js';
|
|
20
23
|
import { normalizeFollowups } from './followup-utils.js';
|
|
21
|
-
import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool,
|
|
24
|
+
import { assessEditActivity, detectUnverifiedEditClaim, isVerificationTool, verificationNudgeFor, THINK_ONLY_ESCALATION, AUTHORIZED_WORK_NUDGE, deliverableNudge, } from './edit-verification.js';
|
|
22
25
|
import { effectiveToolJsonSchemas, coreToolJsonSchemas, isToolEnabled, toolsetForTool } from './toolsets.js';
|
|
23
26
|
import { appendToolArtifact } from './artifact-append.js';
|
|
24
27
|
import { logger } from '../utils/logger.js';
|
|
25
|
-
import { toUserFacingGenerationError } from '../inference/tool-call-utils.js';
|
|
28
|
+
import { toUserFacingGenerationError, TEXTUAL_TOOL_CALL_HINT, isTextualFollowupsPayload, followupEntriesFromPayload, stripTrailingFollowupsHeader, } from '../inference/tool-call-utils.js';
|
|
26
29
|
// ─── P3c — tool-fallback hints (switch tools when one fails) ────────────────
|
|
27
30
|
// The ask: *"if one tool fails you switch to another and explore parallel
|
|
28
31
|
// ways"* (round-2 row 25). Today the raw `Error: …` text is fed back and a
|
|
@@ -115,7 +118,8 @@ export function isThinkOnlyResponse(content) {
|
|
|
115
118
|
}
|
|
116
119
|
/**
|
|
117
120
|
* Extract JSON fallback tool calls from model text:
|
|
118
|
-
* `{"tool":"name","arguments":{...}}` blocks, possibly fenced or multiple
|
|
121
|
+
* `{"tool":"name","arguments":{...}}` blocks, possibly fenced or multiple —
|
|
122
|
+
* plus the name-keyed `{"suggest_followups":[…]}` shape models actually write.
|
|
119
123
|
* Uses brace-matching (string-aware) so nested argument objects parse
|
|
120
124
|
* correctly. Returns the cleaned content (blocks stripped) + parsed calls.
|
|
121
125
|
*/
|
|
@@ -160,11 +164,59 @@ export function extractFallbackToolCalls(content) {
|
|
|
160
164
|
// Unparseable block — dropped from the answer, no tool call.
|
|
161
165
|
}
|
|
162
166
|
}
|
|
167
|
+
// ── Pass 2: our tool's ARGUMENTS keyed by our own tool NAME ─────────────
|
|
168
|
+
// {"suggest_followups":[{"label":"…","prompt":"…"}, …]}
|
|
169
|
+
// This is the shape models ACTUALLY hand-write: across 116 stored assistant
|
|
170
|
+
// turns the canonical `{"tool":"…"}` form appeared ZERO times and this one
|
|
171
|
+
// 14. Pass 1 could not see it at all, so the block was delivered verbatim to
|
|
172
|
+
// the reader AND the suggestions were thrown away (no chips, no menu).
|
|
173
|
+
//
|
|
174
|
+
// The block is consumed in exactly two cases: it parses to OUR payload (the
|
|
175
|
+
// shared predicate — the same one the strip uses, so the two can never
|
|
176
|
+
// disagree), or it does not parse at all (truncated scaffolding is never
|
|
177
|
+
// prose). A value that parses cleanly to something demonstrably NOT the
|
|
178
|
+
// contract — `{"suggest_followups": "a note"}` — is left untouched, so a
|
|
179
|
+
// user's own JSON still survives.
|
|
180
|
+
const namedAt = /\(?\s*\{\s*"suggest_followups"\s*:/g;
|
|
181
|
+
let named;
|
|
182
|
+
while ((named = namedAt.exec(cleaned)) !== null) {
|
|
183
|
+
const end = findMatchingBrace(cleaned, named.index);
|
|
184
|
+
if (end === -1) {
|
|
185
|
+
// Cut off mid-call: everything from the marker on is scaffolding.
|
|
186
|
+
cleaned = cleaned.slice(0, named.index);
|
|
187
|
+
strippedAny = true;
|
|
188
|
+
break;
|
|
189
|
+
}
|
|
190
|
+
const block = cleaned.slice(named.index, end + 1);
|
|
191
|
+
let parsed = null;
|
|
192
|
+
let unparseable = false;
|
|
193
|
+
try {
|
|
194
|
+
parsed = JSON.parse(block);
|
|
195
|
+
}
|
|
196
|
+
catch {
|
|
197
|
+
unparseable = true;
|
|
198
|
+
}
|
|
199
|
+
if (!unparseable && !isTextualFollowupsPayload(parsed))
|
|
200
|
+
continue;
|
|
201
|
+
cleaned = cleaned.slice(0, named.index) + cleaned.slice(end + 1);
|
|
202
|
+
namedAt.lastIndex = named.index;
|
|
203
|
+
strippedAny = true;
|
|
204
|
+
const entries = followupEntriesFromPayload(parsed);
|
|
205
|
+
if (entries) {
|
|
206
|
+
calls.push({
|
|
207
|
+
id: `call_${calls.length + 1}`,
|
|
208
|
+
name: 'suggest_followups',
|
|
209
|
+
arguments: { followups: entries },
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
}
|
|
163
213
|
if (!strippedAny)
|
|
164
214
|
return { text: content, calls };
|
|
165
215
|
// Remove empty fenced blocks left behind when a fenced JSON block's BODY was
|
|
166
|
-
// the tool call (e.g. "```json\n\n```")
|
|
167
|
-
|
|
216
|
+
// the tool call (e.g. "```json\n\n```"), plus any caption/separator the model
|
|
217
|
+
// put in FRONT of the block. Without the second step the delivered answer
|
|
218
|
+
// ends on a dangling `---` — the visible half of the leak this recovers from.
|
|
219
|
+
const text = stripTrailingFollowupsHeader(cleaned.replace(/\n?```[a-z]*\s*\n?\s*```\s*/gi, '\n')).trim();
|
|
168
220
|
return { text, calls };
|
|
169
221
|
}
|
|
170
222
|
/**
|
|
@@ -256,6 +308,16 @@ const MAX_PARALLEL_READS = 4;
|
|
|
256
308
|
// half-done and resumable in-context. These defaults resume the SAME turn a
|
|
257
309
|
// bounded number of times — the model keeps its thread, completed tool calls
|
|
258
310
|
// are never re-run, and the loop can never spin forever.
|
|
311
|
+
/**
|
|
312
|
+
* Consecutive REASONING-ONLY steps the loop tolerates before it escalates, and
|
|
313
|
+
* then one more before it ends the turn.
|
|
314
|
+
*
|
|
315
|
+
* Small on purpose. A `<think>`-then-answer model needs a step or two, so the
|
|
316
|
+
* limit cannot be zero — but unbounded, this path is a spin: two live eval runs
|
|
317
|
+
* produced 31 reasoning-only steps, one tool call, and a 0% score on a task they
|
|
318
|
+
* were well able to do (see `THINK_ONLY_ESCALATION`).
|
|
319
|
+
*/
|
|
320
|
+
export const MAX_THINK_CONTINUES = 3;
|
|
259
321
|
/** Default continuations granted per turn when the option is omitted. */
|
|
260
322
|
export const DEFAULT_MAX_CONTINUATIONS = 2;
|
|
261
323
|
/** Default extra steps granted per continuation. */
|
|
@@ -295,6 +357,30 @@ async function runToolLoopInner(opts, progress) {
|
|
|
295
357
|
// G13 — bounded "the request already authorized this" nudges (see
|
|
296
358
|
// detectPermissionSeeking).
|
|
297
359
|
let permissionNudges = 0;
|
|
360
|
+
// Stage 2 — the repetition nudge's own bound (see the block after the
|
|
361
|
+
// permission nudge). Separate from `permissionNudges` on purpose: that one
|
|
362
|
+
// asks the model to PROCEED, this one asks it to stop REPEATING, and the two
|
|
363
|
+
// fire on independent evidence.
|
|
364
|
+
let repeatNudges = 0;
|
|
365
|
+
// Bounded think-only continuation (see THINK_ONLY_ESCALATION). Counts the
|
|
366
|
+
// consecutive reasoning-only steps so the loop cannot spin on them.
|
|
367
|
+
let thinkContinues = 0;
|
|
368
|
+
// G13b — bounded "the request asked for a file and none was written" nudges
|
|
369
|
+
// (see wantsAuthoredArtifact).
|
|
370
|
+
let deliverableNudges = 0;
|
|
371
|
+
/**
|
|
372
|
+
* G18 — the sink, wrapped so an observability failure can never become a
|
|
373
|
+
* turn failure (a recorder that throws is a bug in the instrument, not in
|
|
374
|
+
* the work being observed).
|
|
375
|
+
*/
|
|
376
|
+
const traceEvent = (event) => {
|
|
377
|
+
try {
|
|
378
|
+
opts.onTraceEvent?.(event);
|
|
379
|
+
}
|
|
380
|
+
catch {
|
|
381
|
+
// Best-effort.
|
|
382
|
+
}
|
|
383
|
+
};
|
|
298
384
|
// The EFFECTIVE bound: starts at maxSteps and is extended (never beyond the
|
|
299
385
|
// continuation budget) so a turn that still has work can finish.
|
|
300
386
|
let stepLimit = maxSteps;
|
|
@@ -325,6 +411,45 @@ async function runToolLoopInner(opts, progress) {
|
|
|
325
411
|
// gate exactly as it was.
|
|
326
412
|
const requestText = lastUserText(opts.messages);
|
|
327
413
|
const authorization = requestAuthorizesWrites(requestText);
|
|
414
|
+
// ── INTENT ENVELOPE (the durable grant) ──────────────────────────────────
|
|
415
|
+
// `writesAuthorized` above answers "does the LAST message ask for files?" —
|
|
416
|
+
// once per turn, and forgotten at the turn boundary. That is why permission
|
|
417
|
+
// was re-asked for every operation and why an approval given minutes ago was
|
|
418
|
+
// worth nothing on the next turn. The envelope is the same verdict as a
|
|
419
|
+
// DURABLE, SCOPED object, keyed to the conversation (the dashboard console
|
|
420
|
+
// keeps one plan store per session and re-injects it every turn; the CLI keeps
|
|
421
|
+
// one per ChatCommand), so an approval outlives the turn that granted it.
|
|
422
|
+
//
|
|
423
|
+
// Grant order is deliberate: an APPROVED PLAN is the high-trust path (the user
|
|
424
|
+
// saw the plan and said yes), then a fresh directive request, then whatever is
|
|
425
|
+
// still live for this conversation.
|
|
426
|
+
const envelopeKey = isEnvelopeKey(context.planStore) ? context.planStore : undefined;
|
|
427
|
+
let envelope = getEnvelope(envelopeKey);
|
|
428
|
+
if (envelopeKey) {
|
|
429
|
+
const plan = context.planStore?.snapshot?.() ?? null;
|
|
430
|
+
if (plan && isAffirmativeReply(requestText)) {
|
|
431
|
+
envelope = envelopeFromPlan(plan.goal);
|
|
432
|
+
grantEnvelope(envelopeKey, envelope);
|
|
433
|
+
deps.onEvent?.(' ✅ Plan approved — running it under one grant; no per-step permission.');
|
|
434
|
+
}
|
|
435
|
+
else {
|
|
436
|
+
const fresh = envelopeFromRequest(requestText);
|
|
437
|
+
if (fresh) {
|
|
438
|
+
envelope = fresh;
|
|
439
|
+
grantEnvelope(envelopeKey, fresh);
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
// ── RUN TRACE (the loop's representation of its own behaviour) ────────────
|
|
444
|
+
// The envelope above answers "what may I do"; the trace answers "what have I
|
|
445
|
+
// been DOING" — the question nothing in the architecture could answer, which
|
|
446
|
+
// is why it could not notice its own loop and could not respond to a user
|
|
447
|
+
// asking about one. Keyed to the conversation like the envelope, with a
|
|
448
|
+
// per-turn fallback when there is no session handle so repetition WITHIN a
|
|
449
|
+
// single turn is still caught.
|
|
450
|
+
const traceKey = isTraceKey(context.planStore) ? context.planStore : undefined;
|
|
451
|
+
const runTrace = traceKey ? runTraceFor(traceKey) : new RunTrace();
|
|
452
|
+
progress.runTrace = runTrace;
|
|
328
453
|
const ctx = {
|
|
329
454
|
...context,
|
|
330
455
|
followups: context.followups || sink,
|
|
@@ -334,6 +459,11 @@ async function runToolLoopInner(opts, progress) {
|
|
|
334
459
|
// ask for a commit? resolve to this CLI command?), and a single boolean
|
|
335
460
|
// computed for a different question cannot answer any of them.
|
|
336
461
|
authorizationRequest: requestText,
|
|
462
|
+
// The durable grant every gate consults FIRST (see learning/intent-envelope.ts).
|
|
463
|
+
envelope,
|
|
464
|
+
// The run's own behaviour, so a gate can refuse to RE-ASK a question the
|
|
465
|
+
// user already answered (see learning/run-trace.ts).
|
|
466
|
+
runTrace,
|
|
337
467
|
};
|
|
338
468
|
// Resolve the tool set — stable JSON schemas for every native step
|
|
339
469
|
// (derived from the registry's zod schemas, single source of truth).
|
|
@@ -389,6 +519,28 @@ async function runToolLoopInner(opts, progress) {
|
|
|
389
519
|
return added;
|
|
390
520
|
};
|
|
391
521
|
const thread = [...messages];
|
|
522
|
+
// ── Stage 2: answer a question ABOUT the run from the run ────────────────
|
|
523
|
+
// The live audit's worst moment was the user asking "why are you asking me
|
|
524
|
+
// this again and again?" and the turn replying with an edit plan — not through
|
|
525
|
+
// indifference, but because nothing held the fact that it had asked four
|
|
526
|
+
// times. Handing the trace in makes the honest answer possible; the framing
|
|
527
|
+
// line keeps it from being mistaken for a plan.
|
|
528
|
+
if (detectProcessComplaint(requestText)) {
|
|
529
|
+
deps.onEvent?.(' 🪞 The user asked about your own behaviour — answering from the run trace.');
|
|
530
|
+
traceEvent({
|
|
531
|
+
kind: 'gate',
|
|
532
|
+
gate: 'repeat',
|
|
533
|
+
summary: `the user asked about the agent's own behaviour — trace provided (${runTrace.countAsks()} ask(s), ${runTrace.repeatedAskCount()} repeated)`,
|
|
534
|
+
});
|
|
535
|
+
thread.push({
|
|
536
|
+
role: 'system',
|
|
537
|
+
content: `${runTrace.selfReport()}\n\n` +
|
|
538
|
+
'The user is asking about THIS behaviour, not about the code. Answer them directly: ' +
|
|
539
|
+
'state what you did (how many questions you asked, which one you repeated, and what they ' +
|
|
540
|
+
'answered), acknowledge the repetition plainly if there was one, and say what you will do ' +
|
|
541
|
+
'instead. Do not answer a question about your own process with a plan for the work.',
|
|
542
|
+
});
|
|
543
|
+
}
|
|
392
544
|
const budgetChars = opts.threadBudgetChars ?? DEFAULT_THREAD_BUDGET_CHARS;
|
|
393
545
|
// P3d — per-turn parallel suggester: after 2+ successful independent gather
|
|
394
546
|
// steps, one advisory delegate suggestion fires (bounded, deterministic).
|
|
@@ -525,7 +677,7 @@ async function runToolLoopInner(opts, progress) {
|
|
|
525
677
|
// console, gateway) recovers the calls, and strip the block from the
|
|
526
678
|
// visible answer either way: a raw `{"tool":…}` block must never be the
|
|
527
679
|
// answer. Idempotent with the JSON transport (which already strips it).
|
|
528
|
-
if (response.toolCalls.length === 0 && response.content
|
|
680
|
+
if (response.toolCalls.length === 0 && TEXTUAL_TOOL_CALL_HINT.test(response.content)) {
|
|
529
681
|
const extracted = extractFallbackToolCalls(response.content);
|
|
530
682
|
if (extracted.text !== response.content) {
|
|
531
683
|
response.content = extracted.text;
|
|
@@ -557,8 +709,63 @@ async function runToolLoopInner(opts, progress) {
|
|
|
557
709
|
if (deps.isThinkOnly ? deps.isThinkOnly(response.content) : isThinkOnlyResponse(response.content)) {
|
|
558
710
|
// Feed an empty assistant step so the model continues in-context.
|
|
559
711
|
thread.push({ role: 'assistant', content: response.content });
|
|
560
|
-
|
|
561
|
-
|
|
712
|
+
thinkContinues += 1;
|
|
713
|
+
// AN EMPTY RESPONSE IS A FAILURE, NOT REASONING. `isThinkOnlyResponse('')`
|
|
714
|
+
// returns true, so a provider returning nothing (the loop arm falls over
|
|
715
|
+
// to `local/gpt-oss:120b-cloud`, which reliably returns length 0) was
|
|
716
|
+
// reported as "model reasoning… (continuing)" — the system's own evidence
|
|
717
|
+
// saying the agent was thinking when it was being handed nothing. That
|
|
718
|
+
// is the exact class of defect this work exists to remove, so the two
|
|
719
|
+
// cases are named apart even though they are bounded together.
|
|
720
|
+
const emptyResponse = response.content.trim().length === 0;
|
|
721
|
+
// BOUNDED (see THINK_ONLY_ESCALATION). Continuing on reasoning is right
|
|
722
|
+
// for a `<think>`-then-answer model and catastrophic without a limit:
|
|
723
|
+
// two live eval runs produced 31 reasoning-only steps, one tool call and
|
|
724
|
+
// a 0% score, printing "model reasoning… (continuing)" the whole way.
|
|
725
|
+
if (thinkContinues <= MAX_THINK_CONTINUES) {
|
|
726
|
+
deps.onEvent?.(emptyResponse
|
|
727
|
+
? ` ⚠️ the provider returned an EMPTY response (${thinkContinues}/${MAX_THINK_CONTINUES}) — retrying the step.`
|
|
728
|
+
: ` 🧠 model reasoning… (continuing ${thinkContinues}/${MAX_THINK_CONTINUES})`);
|
|
729
|
+
traceEvent({
|
|
730
|
+
kind: 'decision',
|
|
731
|
+
summary: emptyResponse
|
|
732
|
+
? 'the provider returned an empty response — no answer text and no tool call; the step was retried'
|
|
733
|
+
: 'the model replied with its own reasoning instead of an answer — the step continued',
|
|
734
|
+
});
|
|
735
|
+
continue;
|
|
736
|
+
}
|
|
737
|
+
if (thinkContinues === MAX_THINK_CONTINUES + 1) {
|
|
738
|
+
// One escalation: it has no answer yet, so tell it to act or answer.
|
|
739
|
+
stepLimit += 1;
|
|
740
|
+
deps.onEvent?.(emptyResponse
|
|
741
|
+
? ' ⚠️ The provider keeps returning empty responses — asking for one real step.'
|
|
742
|
+
: ' 🧠 Reasoning only, repeatedly — telling the model to act or answer now.');
|
|
743
|
+
traceEvent({
|
|
744
|
+
kind: 'gate',
|
|
745
|
+
gate: 'repeat',
|
|
746
|
+
summary: emptyResponse
|
|
747
|
+
? `the provider returned an empty response ${thinkContinues} times — one bounded escalation for a real step`
|
|
748
|
+
: `the model produced reasoning-only output ${thinkContinues} times with no tool call and no answer — ` +
|
|
749
|
+
'one bounded escalation to act or answer',
|
|
750
|
+
});
|
|
751
|
+
thread.push({ role: 'user', content: THINK_ONLY_ESCALATION });
|
|
752
|
+
continue;
|
|
753
|
+
}
|
|
754
|
+
// Still spinning after the escalation: END the turn rather than burn the
|
|
755
|
+
// rest of the budget. `bounded` is set so every surface reads this as
|
|
756
|
+
// "stopped before finishing", never as a completed answer.
|
|
757
|
+
bounded = true;
|
|
758
|
+
deps.onEvent?.(emptyResponse
|
|
759
|
+
? ' ⚠️ Empty responses kept coming — ending the turn instead of spinning. This is a PROVIDER failure, not the agent thinking.'
|
|
760
|
+
: ' 🧠 Reasoning-only output kept repeating — ending the turn instead of spinning.');
|
|
761
|
+
traceEvent({
|
|
762
|
+
kind: 'gate',
|
|
763
|
+
gate: 'repeat',
|
|
764
|
+
summary: emptyResponse
|
|
765
|
+
? `the provider returned ${thinkContinues} consecutive empty responses — the turn ended; this is a transport failure, not agent stuckness`
|
|
766
|
+
: `reasoning-only output repeated ${thinkContinues} times after an escalation — the turn ended to avoid an unbounded spin`,
|
|
767
|
+
});
|
|
768
|
+
break;
|
|
562
769
|
}
|
|
563
770
|
// ── Dangling-promise nudge (bounded, once) ──────────────────────────
|
|
564
771
|
// The model closed the turn announcing what it is ABOUT to do ("I will
|
|
@@ -576,6 +783,11 @@ async function runToolLoopInner(opts, progress) {
|
|
|
576
783
|
intentNudges += 1;
|
|
577
784
|
stepLimit += 1;
|
|
578
785
|
deps.onEvent?.(' 🔁 Answer announced an action but performed none — asking the model to carry it out.');
|
|
786
|
+
traceEvent({
|
|
787
|
+
kind: 'gate',
|
|
788
|
+
gate: 'promise',
|
|
789
|
+
summary: 'the turn closed on an announced action with nothing performed — one bounded nudge to carry it out',
|
|
790
|
+
});
|
|
579
791
|
// The promise is NOT a candidate answer: drop it from the
|
|
580
792
|
// longest-substantive memory so the post-nudge answer (or result) is
|
|
581
793
|
// what the turn delivers. Safe because the gate only fires when no
|
|
@@ -608,6 +820,11 @@ async function runToolLoopInner(opts, progress) {
|
|
|
608
820
|
permissionNudges += 1;
|
|
609
821
|
stepLimit += 1;
|
|
610
822
|
deps.onEvent?.(' 🤖 Turn ended asking permission for work the request already authorized — telling the model to proceed.');
|
|
823
|
+
traceEvent({
|
|
824
|
+
kind: 'gate',
|
|
825
|
+
gate: 'permission',
|
|
826
|
+
summary: `the turn asked permission for authorized work (${authorization.reason}) — one bounded nudge to proceed`,
|
|
827
|
+
});
|
|
611
828
|
// The question must not become the delivered answer. When nothing was
|
|
612
829
|
// done this turn the whole step is the question, so drop it. When real
|
|
613
830
|
// work WAS done, keep the work and cut only the trailing question —
|
|
@@ -623,12 +840,65 @@ async function runToolLoopInner(opts, progress) {
|
|
|
623
840
|
thread.push({ role: 'user', content: AUTHORIZED_WORK_NUDGE });
|
|
624
841
|
continue;
|
|
625
842
|
}
|
|
843
|
+
// ── Stage 2 — REPETITION nudge (bounded, once) ───────────────────────
|
|
844
|
+
// The permission nudge above fires only when the request AUTHORIZED the
|
|
845
|
+
// work, and it asks the model to PROCEED. A repeated question is a defect
|
|
846
|
+
// on its own terms: the user already answered it, whatever their latest
|
|
847
|
+
// message authorized — which is precisely the case a live turn fell
|
|
848
|
+
// through, because the complaint about repeated questions was itself the
|
|
849
|
+
// message that failed to authorize. This nudge therefore does NOT depend
|
|
850
|
+
// on authorization, and deliberately does NOT say "proceed" (unsafe when
|
|
851
|
+
// the answer was no) — it says the answer is in hand, so act on it or say
|
|
852
|
+
// what blocks you. The evidence is the RUN TRACE, not the current text.
|
|
853
|
+
if (repeatNudges < 1 &&
|
|
854
|
+
schemas.length > 0 &&
|
|
855
|
+
detectPermissionSeeking(response.content.length >= lastContent.length ? response.content : lastContent) &&
|
|
856
|
+
runTrace.priorAskMatches(response.content.length >= lastContent.length ? response.content : lastContent).length > 0) {
|
|
857
|
+
const closing = response.content.length >= lastContent.length ? response.content : lastContent;
|
|
858
|
+
repeatNudges += 1;
|
|
859
|
+
stepLimit += 1;
|
|
860
|
+
deps.onEvent?.(' 🔁 The turn asked a question the user has already answered — telling the model to act on the answer.');
|
|
861
|
+
traceEvent({
|
|
862
|
+
kind: 'gate',
|
|
863
|
+
gate: 'repeat',
|
|
864
|
+
summary: 'the turn repeated a question already asked and answered in this conversation — one bounded nudge to use the answer',
|
|
865
|
+
});
|
|
866
|
+
if (progress.successfulToolCalls.length === 0) {
|
|
867
|
+
lastContent = '';
|
|
868
|
+
}
|
|
869
|
+
else {
|
|
870
|
+
lastContent = stripTrailingPermissionSeek(closing);
|
|
871
|
+
}
|
|
872
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
873
|
+
thread.push({ role: 'user', content: repeatNudge(closing, runTrace.priorAnswer(closing)) });
|
|
874
|
+
continue;
|
|
875
|
+
}
|
|
626
876
|
// S1 (both exits): the MOST SUBSTANTIVE content seen wins here too —
|
|
627
877
|
// a short closing step ("Sent it to her! ✅") with no tool calls must
|
|
628
878
|
// not clobber the deliverable (poem/essay) the model composed in an
|
|
629
879
|
// earlier step alongside a real tool call. Same rule as the
|
|
630
880
|
// suggest_followups exit below.
|
|
631
881
|
//
|
|
882
|
+
// G13b — DELIVERABLE GATE (no-tools path, the one a story ask actually
|
|
883
|
+
// takes). Same rule and same bounded budget as the concluding path BELOW,
|
|
884
|
+
// placed before it so the first ending that can satisfy the request gets
|
|
885
|
+
// the nudge, never both.
|
|
886
|
+
if (deliverableNudges < 1 && deliverableGateApplies(opts, response.content.length >= lastContent.length ? response.content : lastContent, progress, schemas.length, requestText)) {
|
|
887
|
+
deliverableNudges += 1;
|
|
888
|
+
stepLimit += 1;
|
|
889
|
+
deps.onEvent?.(' 📄 The request asked for a file and none was written — asking the model to produce it.');
|
|
890
|
+
traceEvent({
|
|
891
|
+
kind: 'gate',
|
|
892
|
+
gate: 'deliverable',
|
|
893
|
+
summary: authorization.requestedPath
|
|
894
|
+
? `an authored deliverable was requested at ${authorization.requestedPath} and no file was written — one bounded nudge to produce it`
|
|
895
|
+
: 'an authored deliverable was requested and no file was written — one bounded nudge to produce it',
|
|
896
|
+
});
|
|
897
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
898
|
+
thread.push({ role: 'user', content: deliverableNudge(authorization.requestedPath) });
|
|
899
|
+
continue;
|
|
900
|
+
}
|
|
901
|
+
//
|
|
632
902
|
// G1 — VERIFICATION GATE: before the turn can end, if it MUTATED the
|
|
633
903
|
// workspace and ran nothing that observed the result, spend ONE bounded
|
|
634
904
|
// nudge asking for the check. The gate is deliberately a nudge, not a
|
|
@@ -643,8 +913,20 @@ async function runToolLoopInner(opts, progress) {
|
|
|
643
913
|
verificationNudges += 1;
|
|
644
914
|
stepLimit += 1;
|
|
645
915
|
deps.onEvent?.(' 🔎 Files were changed but nothing verified them — asking the model to run a check.');
|
|
916
|
+
traceEvent({
|
|
917
|
+
kind: 'gate',
|
|
918
|
+
gate: 'verification',
|
|
919
|
+
summary: 'the turn mutated the workspace and nothing observed the result — one bounded nudge to verify',
|
|
920
|
+
});
|
|
646
921
|
thread.push({ role: 'assistant', content: response.content });
|
|
647
|
-
|
|
922
|
+
// Stage 3 — name THIS project's strongest check instead of describing a
|
|
923
|
+
// preference order. A live turn reached for `node -c` because the nudge
|
|
924
|
+
// left the choice to the model; `verificationNudgeFor` reads the
|
|
925
|
+
// workspace and asks for the real command, so there is nothing to guess.
|
|
926
|
+
thread.push({
|
|
927
|
+
role: 'user',
|
|
928
|
+
content: verificationNudgeFor(ctx.cwd ?? process.cwd(), progress.mutatedPaths),
|
|
929
|
+
});
|
|
648
930
|
continue;
|
|
649
931
|
}
|
|
650
932
|
return {
|
|
@@ -721,6 +1003,9 @@ async function runToolLoopInner(opts, progress) {
|
|
|
721
1003
|
return { call };
|
|
722
1004
|
});
|
|
723
1005
|
// ── Phase 2 — execute (read-only runs fan out; everything else serial) ──
|
|
1006
|
+
/** Wall-clock per call id, so the G18 event can carry a real duration
|
|
1007
|
+
* without re-timing (the emit below is the only place that knows it). */
|
|
1008
|
+
const toolDurations = new Map();
|
|
724
1009
|
const runOne = async (plan) => {
|
|
725
1010
|
const { call } = plan;
|
|
726
1011
|
if (plan.refuse !== undefined)
|
|
@@ -771,6 +1056,9 @@ async function runToolLoopInner(opts, progress) {
|
|
|
771
1056
|
});
|
|
772
1057
|
return `Error: ${message}`;
|
|
773
1058
|
}
|
|
1059
|
+
finally {
|
|
1060
|
+
toolDurations.set(call.id, Date.now() - startedAt);
|
|
1061
|
+
}
|
|
774
1062
|
};
|
|
775
1063
|
const executed = new Array(plans.length).fill('');
|
|
776
1064
|
for (let i = 0; i < plans.length;) {
|
|
@@ -816,7 +1104,17 @@ async function runToolLoopInner(opts, progress) {
|
|
|
816
1104
|
// not a success, and a delivery tool only counts when it reported `✅`).
|
|
817
1105
|
// `toolCallsRun` stays the attempted list for telemetry; these two drive
|
|
818
1106
|
// the honesty flag and the trace outcome.
|
|
819
|
-
|
|
1107
|
+
// A DECLINED call is not a success, whatever prefix it used. Found live on
|
|
1108
|
+
// the first G18 verification run: `write_file` refused an absolute path
|
|
1109
|
+
// that escaped the workspace and returned "… escapes the workspace … —
|
|
1110
|
+
// denied" with NO `Error:` prefix, so the loop counted it as a call that
|
|
1111
|
+
// RAN, pushed it to `successfulToolCalls`, and recorded it in the trace as
|
|
1112
|
+
// "write_file ran". That is the exact shape G18 exists to eliminate — an
|
|
1113
|
+
// outcome reading the same as its opposite — so the refusal classifier is
|
|
1114
|
+
// now the authority for both the event KIND and the success verdict.
|
|
1115
|
+
const refusal = classifyToolRefusal(rawResult);
|
|
1116
|
+
const ranOk = refusal === null &&
|
|
1117
|
+
!rawResult.startsWith('Error:') &&
|
|
820
1118
|
(!DELIVERY_TOOL_NAMES.has(call.name) || deliveryResultSucceeded(rawResult));
|
|
821
1119
|
if (ranOk)
|
|
822
1120
|
progress.successfulToolCalls.push(call.name);
|
|
@@ -830,13 +1128,25 @@ async function runToolLoopInner(opts, progress) {
|
|
|
830
1128
|
if (call.name === 'edit_file' || call.name === 'write_file') {
|
|
831
1129
|
const a = call.arguments;
|
|
832
1130
|
const p = a?.path ?? a?.file_path ?? a?.file;
|
|
833
|
-
if (typeof p === 'string' && p)
|
|
1131
|
+
if (typeof p === 'string' && p) {
|
|
834
1132
|
progress.mutatedPaths.push(p);
|
|
1133
|
+
// Stage 2 — the run's own record of what it CHANGED, so a self-report
|
|
1134
|
+
// can answer "what have you been doing" from data rather than from a
|
|
1135
|
+
// re-read of the transcript.
|
|
1136
|
+
runTrace.recordMutation(call.name, p);
|
|
1137
|
+
}
|
|
835
1138
|
}
|
|
836
1139
|
else if (isVerificationTool(call.name)) {
|
|
837
1140
|
progress.verificationEvidence.push({ tool: call.name, args: call.arguments, result: rawResult });
|
|
838
1141
|
}
|
|
839
1142
|
}
|
|
1143
|
+
else if (refusal !== null || rawResult.startsWith('Error:')) {
|
|
1144
|
+
// Stage 2 — the other half of noticing a loop: the SAME refusal, twice.
|
|
1145
|
+
// The run records it with its reason, so a later step (or a self-report)
|
|
1146
|
+
// can see "blocked twice, identically" instead of rediscovering it — and
|
|
1147
|
+
// so the repetition gate has a refusal signal as well as an ask signal.
|
|
1148
|
+
runTrace.recordRefusal(call.name, refusal?.gate ?? 'error', rawResult);
|
|
1149
|
+
}
|
|
840
1150
|
let resultText = executed[i];
|
|
841
1151
|
// P3c — on error/denial, append the deterministic fallback hint for
|
|
842
1152
|
// this tool (advisory — the model still decides; never on success).
|
|
@@ -849,6 +1159,24 @@ async function runToolLoopInner(opts, progress) {
|
|
|
849
1159
|
if (parallelTip)
|
|
850
1160
|
resultText = `${resultText}\n\n${parallelTip}`;
|
|
851
1161
|
delivered[i] = resultText;
|
|
1162
|
+
// G18 — record what this call DID, in the model's own call order, with the
|
|
1163
|
+
// evidence a later reader needs: the args the gate saw, the first line of
|
|
1164
|
+
// the result, the verdict, and the wall-clock. A declined call is recorded
|
|
1165
|
+
// as a REFUSAL with its cause, so "nothing happened here" can never again
|
|
1166
|
+
// read the same as "nothing was refused" (see classifyToolRefusal).
|
|
1167
|
+
// `rawResult` is the pre-decoration text — hints and tips are the loop's
|
|
1168
|
+
// own additions and must not be mistaken for what the tool returned.
|
|
1169
|
+
const durationMs = toolDurations.get(call.id);
|
|
1170
|
+
traceEvent({
|
|
1171
|
+
kind: ranOk ? 'tool' : 'refusal',
|
|
1172
|
+
tool: call.name,
|
|
1173
|
+
...(refusal?.gate ? { gate: refusal.gate } : {}),
|
|
1174
|
+
summary: ranOk ? `${call.name} ran` : refusal?.summary ?? `${call.name} returned an error`,
|
|
1175
|
+
args: summarizeArgs(call.arguments),
|
|
1176
|
+
result: previewToolResult(rawResult),
|
|
1177
|
+
ok: ranOk,
|
|
1178
|
+
...(durationMs !== undefined ? { durationMs } : {}),
|
|
1179
|
+
});
|
|
852
1180
|
thread.push({ role: 'tool', toolCallId: call.id, content: resultText });
|
|
853
1181
|
}
|
|
854
1182
|
// Tiered exposure: after EVERY executed tool call, union any newly
|
|
@@ -923,8 +1251,38 @@ async function runToolLoopInner(opts, progress) {
|
|
|
923
1251
|
verificationNudges += 1;
|
|
924
1252
|
stepLimit += 1;
|
|
925
1253
|
deps.onEvent?.(' 🔎 Files were changed but nothing verified them — asking the model to run a check.');
|
|
1254
|
+
traceEvent({
|
|
1255
|
+
kind: 'gate',
|
|
1256
|
+
gate: 'verification',
|
|
1257
|
+
summary: 'the turn mutated the workspace and nothing observed the result — one bounded nudge to verify',
|
|
1258
|
+
});
|
|
1259
|
+
thread.push({ role: 'assistant', content: response.content });
|
|
1260
|
+
// Stage 3 — name THIS project's strongest check instead of describing a
|
|
1261
|
+
// preference order. A live turn reached for `node -c` because the nudge
|
|
1262
|
+
// left the choice to the model; `verificationNudgeFor` reads the
|
|
1263
|
+
// workspace and asks for the real command, so there is nothing to guess.
|
|
1264
|
+
thread.push({
|
|
1265
|
+
role: 'user',
|
|
1266
|
+
content: verificationNudgeFor(ctx.cwd ?? process.cwd(), progress.mutatedPaths),
|
|
1267
|
+
});
|
|
1268
|
+
continue;
|
|
1269
|
+
}
|
|
1270
|
+
// G13b — DELIVERABLE GATE (concluding path). AFTER the verification gate
|
|
1271
|
+
// on purpose: a turn that wrote nothing has nothing to verify, so the two
|
|
1272
|
+
// cannot both apply, and this order leaves edit turns byte-identical.
|
|
1273
|
+
if (deliverableNudges < 1 && deliverableGateApplies(opts, response.content.length >= lastContent.length ? response.content : lastContent, progress, schemas.length, requestText)) {
|
|
1274
|
+
deliverableNudges += 1;
|
|
1275
|
+
stepLimit += 1;
|
|
1276
|
+
deps.onEvent?.(' 📄 The request asked for a file and none was written — asking the model to produce it.');
|
|
1277
|
+
traceEvent({
|
|
1278
|
+
kind: 'gate',
|
|
1279
|
+
gate: 'deliverable',
|
|
1280
|
+
summary: authorization.requestedPath
|
|
1281
|
+
? `an authored deliverable was requested at ${authorization.requestedPath} and no file was written — one bounded nudge to produce it`
|
|
1282
|
+
: 'an authored deliverable was requested and no file was written — one bounded nudge to produce it',
|
|
1283
|
+
});
|
|
926
1284
|
thread.push({ role: 'assistant', content: response.content });
|
|
927
|
-
thread.push({ role: 'user', content:
|
|
1285
|
+
thread.push({ role: 'user', content: deliverableNudge(authorization.requestedPath) });
|
|
928
1286
|
continue;
|
|
929
1287
|
}
|
|
930
1288
|
const content = response.content.length >= lastContent.length ? response.content : lastContent;
|
|
@@ -942,6 +1300,11 @@ async function runToolLoopInner(opts, progress) {
|
|
|
942
1300
|
// extends the bound otherwise) — the turn is honestly bounded.
|
|
943
1301
|
bounded = true;
|
|
944
1302
|
deps.onEvent?.(` ⚠️ Tool loop reached its ${stepLimit}-step budget — returning the last response.`);
|
|
1303
|
+
traceEvent({
|
|
1304
|
+
kind: 'decision',
|
|
1305
|
+
gate: 'budget',
|
|
1306
|
+
summary: `the step budget (${stepLimit}) was reached — the turn ended on its last response instead of finishing`,
|
|
1307
|
+
});
|
|
945
1308
|
return {
|
|
946
1309
|
content: lastContent || 'I reached my step limit for this request.',
|
|
947
1310
|
followups,
|
|
@@ -1092,6 +1455,8 @@ export async function runToolLoop(opts) {
|
|
|
1092
1455
|
if (!result.cancelled && !result.generationFailed) {
|
|
1093
1456
|
result.successfulToolCalls = [...progress.successfulToolCalls];
|
|
1094
1457
|
result.deliveryConfirmed = progress.deliveryConfirmed;
|
|
1458
|
+
// Stage 2 — the turn's own behaviour, as counts (see learning/run-trace.ts).
|
|
1459
|
+
result.runTrace = progress.runTrace?.snapshot();
|
|
1095
1460
|
// Judge the "I have sent it" claim against what actually DELIVERED, not
|
|
1096
1461
|
// against the attempted tool list — a failed gateway_send must not silence
|
|
1097
1462
|
// the honesty correction.
|
|
@@ -1117,9 +1482,55 @@ export async function runToolLoop(opts) {
|
|
|
1117
1482
|
if (detectUnverifiedEditClaim(result.content, activity.mutations, activity.verifications)) {
|
|
1118
1483
|
result.unverifiedEditClaim = true;
|
|
1119
1484
|
}
|
|
1485
|
+
// G13b — DELIVERABLE honesty, the same way and for the same reason: the
|
|
1486
|
+
// flag is a function of what the turn DID, never of configuration. A turn
|
|
1487
|
+
// that was asked for a file and wrote none must not be readable as
|
|
1488
|
+
// "finished" by any caller, whether or not the gate was enabled.
|
|
1489
|
+
if (!result.cancelled &&
|
|
1490
|
+
progress.mutatedPaths.length === 0 &&
|
|
1491
|
+
!replyAsksTheReader(result.content) &&
|
|
1492
|
+
wantsAuthoredArtifact(lastUserText(opts.messages))) {
|
|
1493
|
+
result.undeliveredArtifact = true;
|
|
1494
|
+
}
|
|
1120
1495
|
}
|
|
1121
1496
|
return result;
|
|
1122
1497
|
}
|
|
1498
|
+
/**
|
|
1499
|
+
* Does the DELIVERABLE gate apply to this turn (G13b)?
|
|
1500
|
+
*
|
|
1501
|
+
* The conjunction is the whole point, and every term is an independent fact:
|
|
1502
|
+
*
|
|
1503
|
+
* - the request asks for an AUTHORED deliverable to be PRODUCED
|
|
1504
|
+
* (`wantsAuthoredArtifact` — the same predicate the engine router uses, so
|
|
1505
|
+
* "what the user asked for" is defined once);
|
|
1506
|
+
* - the turn wrote NOTHING (`mutatedPaths` is the loop's own evidence that a
|
|
1507
|
+
* write landed, so a turn that used a heredoc through `run_terminal` is not
|
|
1508
|
+
* nudged for a file it already wrote);
|
|
1509
|
+
* - tools are actually available (a caller that exposed none cannot comply);
|
|
1510
|
+
* - the turn did not END on a question to the reader (`replyAsksTheReader`).
|
|
1511
|
+
* This is the line between the two failures that look alike: "Do you want me
|
|
1512
|
+
* to create the files?" is a stall, and the permission nudge already settles
|
|
1513
|
+
* it above; "Which of these two titles do you prefer?" is a decision input
|
|
1514
|
+
* the user asked to be consulted on, and bulldozing it would be the manual
|
|
1515
|
+
* -cadence complaint in reverse;
|
|
1516
|
+
* - the bound is unspent, and the caller has not disabled the gate.
|
|
1517
|
+
*
|
|
1518
|
+
* A turn that already wrote something is deliberately out of scope: the gate is
|
|
1519
|
+
* about the artifact never being produced, not about the artifact being short.
|
|
1520
|
+
*/
|
|
1521
|
+
function deliverableGateApplies(opts, content, progress, schemaCount, requestText) {
|
|
1522
|
+
if (opts.requireDeliverable === false)
|
|
1523
|
+
return false;
|
|
1524
|
+
if (schemaCount === 0)
|
|
1525
|
+
return false;
|
|
1526
|
+
if (progress.mutatedPaths.length > 0)
|
|
1527
|
+
return false;
|
|
1528
|
+
if (!requestText.trim())
|
|
1529
|
+
return false;
|
|
1530
|
+
if (replyAsksTheReader(content))
|
|
1531
|
+
return false;
|
|
1532
|
+
return wantsAuthoredArtifact(requestText);
|
|
1533
|
+
}
|
|
1123
1534
|
/**
|
|
1124
1535
|
* The text of the LAST user message in the thread — what the user is actually
|
|
1125
1536
|
* asking for right now, as opposed to the history above it. Used to derive
|
|
@@ -1138,6 +1549,62 @@ function lastUserText(messages) {
|
|
|
1138
1549
|
}
|
|
1139
1550
|
return '';
|
|
1140
1551
|
}
|
|
1552
|
+
/**
|
|
1553
|
+
* One bounded line from a tool result — the EVIDENCE a gate verdict read (G18).
|
|
1554
|
+
*
|
|
1555
|
+
* The first non-empty line, whitespace-collapsed: enough to tell an applied edit
|
|
1556
|
+
* from a declined one ("Error: write_file: … needs explicit confirmation"), a
|
|
1557
|
+
* failing test from a passing one, without pasting a whole file into the trace
|
|
1558
|
+
* store. The store keeps the tail of a long run; a full result per call would
|
|
1559
|
+
* push the interesting end of the turn out of it.
|
|
1560
|
+
*/
|
|
1561
|
+
function previewToolResult(result, max = 200) {
|
|
1562
|
+
const first = (result || '').split(/\r?\n/).find((line) => line.trim() !== '') ?? '';
|
|
1563
|
+
const flat = first.replace(/\s+/g, ' ').trim();
|
|
1564
|
+
return flat.length > max ? `${flat.slice(0, max)}\u2026` : flat;
|
|
1565
|
+
}
|
|
1566
|
+
/**
|
|
1567
|
+
* Classify a DECLINED call's result text (G18).
|
|
1568
|
+
*
|
|
1569
|
+
* The single class this exists for is the confirmation refusal: a tool that
|
|
1570
|
+
* says "call ask_user first, then retry with confirm:true" has not failed — it
|
|
1571
|
+
* has handed the decision to the user, and that is exactly the event the trace
|
|
1572
|
+
* store could not see at all (the G18 audit read "0 refusals" and could not
|
|
1573
|
+
* tell it from "no refusals"). The remaining cases are the loop's own guards,
|
|
1574
|
+
* named so the record distinguishes *why* work did not happen: a guard is not a
|
|
1575
|
+
* provider error — the turn can continue either way, and only the record knows.
|
|
1576
|
+
*
|
|
1577
|
+
* Returns null when the result is an ordinary tool/runtime error, so the caller
|
|
1578
|
+
* keeps its own wording rather than inventing a cause.
|
|
1579
|
+
*/
|
|
1580
|
+
export function classifyToolRefusal(result) {
|
|
1581
|
+
const text = result || '';
|
|
1582
|
+
if (/needs explicit confirmation|requires confirmation|retry with confirm|call ask_user/i.test(text)) {
|
|
1583
|
+
return { gate: 'confirmation', summary: 'declined until the user approves — the gate asked before acting' };
|
|
1584
|
+
}
|
|
1585
|
+
// A BOUNDARY denial is the other refusal that did not read as one. The
|
|
1586
|
+
// workspace guard returns "… escapes the workspace (…) — denied" with no
|
|
1587
|
+
// `Error:` prefix, so it counted as a successful call (found live). Anchored
|
|
1588
|
+
// to the boundary phrasings rather than the bare word "denied", because a
|
|
1589
|
+
// `run_terminal` that reads a log containing "Permission denied" is a
|
|
1590
|
+
// successful call, not a refusal.
|
|
1591
|
+
if (/escapes the workspace|outside the workspace|outside the project|outside this project|not allowed by the workspace/i.test(text)) {
|
|
1592
|
+
return { gate: 'workspace', summary: 'declined — the path is outside the workspace the tools may touch' };
|
|
1593
|
+
}
|
|
1594
|
+
if (/already called|already dispatched/i.test(text)) {
|
|
1595
|
+
return { summary: 'declined by the loop guard — that dispatch already happened this turn' };
|
|
1596
|
+
}
|
|
1597
|
+
if (/^Error: unknown tool/i.test(text)) {
|
|
1598
|
+
return { summary: 'declined — the model called a tool that is not on its surface' };
|
|
1599
|
+
}
|
|
1600
|
+
if (/toolset is not loaded this turn/i.test(text)) {
|
|
1601
|
+
return { summary: 'declined — that tool\'s toolset was never loaded this turn' };
|
|
1602
|
+
}
|
|
1603
|
+
if (/is disabled — its toolset is turned off/i.test(text)) {
|
|
1604
|
+
return { summary: 'declined — the tool is turned off in configuration' };
|
|
1605
|
+
}
|
|
1606
|
+
return null;
|
|
1607
|
+
}
|
|
1141
1608
|
/** Compact argument preview for the event line. */
|
|
1142
1609
|
function summarizeArgs(args) {
|
|
1143
1610
|
const first = Object.entries(args).slice(0, 1)[0];
|