headlesscode 1.0.3 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -56
- package/package.json +1 -1
- package/src/cli.ts +115 -4
- package/src/engine/claims.ts +503 -0
- package/src/engine/events.ts +32 -0
- package/src/engine/lazy-tools.ts +30 -9
- package/src/engine/loop.ts +462 -7
- package/src/engine/prompt.ts +5 -1
- package/src/engine/types.ts +36 -0
- package/src/rsi/archive.ts +129 -0
- package/src/rsi/config.ts +312 -0
- package/src/rsi/controller.ts +268 -0
- package/src/rsi/curriculum.ts +68 -0
- package/src/rsi/evaluator.ts +106 -0
- package/src/rsi/fitness.ts +64 -0
- package/src/rsi/index.ts +16 -0
- package/src/rsi/models.ts +89 -0
- package/src/rsi/mutation.ts +77 -0
- package/src/rsi/reports.ts +47 -0
- package/src/rsi/roles.ts +37 -0
- package/src/rsi/sandbox.ts +10 -0
- package/src/rsi/search.ts +32 -0
- package/src/rsi/selection.ts +132 -0
- package/src/rsi/trajectory.ts +143 -0
- package/src/rsi/types.ts +317 -0
- package/src/rsi/workspace.ts +96 -0
- package/src/tools/executor.ts +79 -3
package/src/engine/loop.ts
CHANGED
|
@@ -62,6 +62,14 @@ import type { MemoryStore, RecallResult } from "../memory/types.js"
|
|
|
62
62
|
import { createCheckpointService, type CheckpointService } from "../checkpoints/service.js"
|
|
63
63
|
import { recordSessionUsage, removeLiveUsage, writeLiveUsage } from "./usage.js"
|
|
64
64
|
import { EventFeed, EVENT_TRUNCATE_CHARS, truncateField } from "./events.js"
|
|
65
|
+
import {
|
|
66
|
+
allClaimsVerified,
|
|
67
|
+
claimLabel,
|
|
68
|
+
extractClaims,
|
|
69
|
+
firstUnverifiedDetail,
|
|
70
|
+
verifyClaims,
|
|
71
|
+
type ClaimVerification,
|
|
72
|
+
} from "./claims.js"
|
|
65
73
|
import { writeSessionReport } from "./reports.js"
|
|
66
74
|
import { Logger } from "./logger.js"
|
|
67
75
|
import { extractEmbeddedToolCall, parseToolCalls } from "./parser.js"
|
|
@@ -89,6 +97,7 @@ import type {
|
|
|
89
97
|
ParsedToolCall,
|
|
90
98
|
SessionResult,
|
|
91
99
|
SessionBudgetUsage,
|
|
100
|
+
SessionCompletionVerification,
|
|
92
101
|
ToolContext,
|
|
93
102
|
ToolResult,
|
|
94
103
|
} from "./types.js"
|
|
@@ -1068,6 +1077,37 @@ export interface HeadlessSessionConfig {
|
|
|
1068
1077
|
* code-mode backend turns this on.
|
|
1069
1078
|
*/
|
|
1070
1079
|
verifyBeforeCompletion?: boolean
|
|
1080
|
+
/**
|
|
1081
|
+
* Evidence-gated completion — the fabrication fix (2026-09-01, see
|
|
1082
|
+
* src/engine/claims.ts). When true, `attempt_completion` is refused
|
|
1083
|
+
* unless every machine-checkable claim its result text makes (a file
|
|
1084
|
+
* exists, a specific command passed, serial markers appear, a PR
|
|
1085
|
+
* exists) is independently verified against ground truth:
|
|
1086
|
+
*
|
|
1087
|
+
* - file claim → fs.stat on the resolved workspace path
|
|
1088
|
+
* - command claim → RE-RUN the exact command (same permission gate as
|
|
1089
|
+
* execute_command), require exit 0
|
|
1090
|
+
* - serial marker → grep the newest build/serial-*.log for the claimed
|
|
1091
|
+
* ordered markers
|
|
1092
|
+
* - PR claim → require real git-history evidence of the number
|
|
1093
|
+
*
|
|
1094
|
+
* Ground truth comes from the filesystem and real re-runs — never from
|
|
1095
|
+
* the model's own prose. This is the structural backstop for the exact
|
|
1096
|
+
* failure the FINAL_REPORT documented (§4): a session claimed "all
|
|
1097
|
+
* three hard gates pass" with a fabricated serial-log excerpt when the
|
|
1098
|
+
* driver was never merged and the claimed Makefile target didn't exist.
|
|
1099
|
+
* Fail-closed: any unverifiable claim defers the completion with a
|
|
1100
|
+
* corrective message naming the specific unverified claim.
|
|
1101
|
+
*
|
|
1102
|
+
* Deliberately separate from verifyBeforeCompletion (which is
|
|
1103
|
+
* token/state-based and only looks at the LAST command): this checks
|
|
1104
|
+
* the CONTENT of the completion's claims against the real world,
|
|
1105
|
+
* independent of what the session did or didn't run before.
|
|
1106
|
+
*
|
|
1107
|
+
* Default false; opt-in via --require-evidence or the local code
|
|
1108
|
+
* backend (see cli.ts).
|
|
1109
|
+
*/
|
|
1110
|
+
evidenceRequiredCompletion?: boolean
|
|
1071
1111
|
/**
|
|
1072
1112
|
* When false, cost is never tracked/accumulated for this session (see
|
|
1073
1113
|
* BudgetTrackerOptions.trackCost) — local-backend sessions have no real
|
|
@@ -1159,6 +1199,24 @@ export interface HeadlessSessionConfig {
|
|
|
1159
1199
|
* the local Ollama code-mode backend turns this on.
|
|
1160
1200
|
*/
|
|
1161
1201
|
guardLargeOverwrites?: boolean
|
|
1202
|
+
/**
|
|
1203
|
+
* read_file/list_files carry a session-scoped "[cache] unchanged, reuse
|
|
1204
|
+
* the earlier result" short-circuit (src/tools/executor.ts) that saves
|
|
1205
|
+
* real, measured token cost against a remote model's per-token bill.
|
|
1206
|
+
* Verified live 2026-09-02 against the local backend: an edit_file
|
|
1207
|
+
* failure told a session to re-read and retry; it DID call read_file
|
|
1208
|
+
* again exactly as instructed, got the cache-hit notice instead of real
|
|
1209
|
+
* content (correct per the mechanism's own design — the safety valve
|
|
1210
|
+
* is "a SECOND consecutive identical call serves real content again"),
|
|
1211
|
+
* never made that second call, and fabricated an attempt_completion
|
|
1212
|
+
* instead. Re-serving a few hundred lines of file content costs a
|
|
1213
|
+
* local session near-nothing (prefill, not generation, against a GPU
|
|
1214
|
+
* with no per-token price) — cheap insurance locally against a much
|
|
1215
|
+
* more expensive failure mode. Default false; the local Ollama code-mode
|
|
1216
|
+
* backend turns this on (same opt-in shape as guardLargeOverwrites
|
|
1217
|
+
* above).
|
|
1218
|
+
*/
|
|
1219
|
+
disableReadFileCache?: boolean
|
|
1162
1220
|
/**
|
|
1163
1221
|
* Phase 3 context condensation: the model's real context window in
|
|
1164
1222
|
* tokens, used to decide WHEN to condense (the last request's real
|
|
@@ -1346,6 +1404,8 @@ export interface ResolvedSessionConfig {
|
|
|
1346
1404
|
patchLocalToolSchemas: boolean
|
|
1347
1405
|
/** See HeadlessSessionConfig.verifyBeforeCompletion (default false). */
|
|
1348
1406
|
verifyBeforeCompletion: boolean
|
|
1407
|
+
/** See HeadlessSessionConfig.evidenceRequiredCompletion (default false). */
|
|
1408
|
+
evidenceRequiredCompletion: boolean
|
|
1349
1409
|
/** See HeadlessSessionConfig.trackCost (default true). */
|
|
1350
1410
|
trackCost: boolean
|
|
1351
1411
|
/** See HeadlessSessionConfig.requireArtifactBeforeCompletion (default false). */
|
|
@@ -1358,6 +1418,8 @@ export interface ResolvedSessionConfig {
|
|
|
1358
1418
|
requireArtifactSections?: string[]
|
|
1359
1419
|
/** See HeadlessSessionConfig.guardLargeOverwrites (default false). */
|
|
1360
1420
|
guardLargeOverwrites: boolean
|
|
1421
|
+
/** See HeadlessSessionConfig.disableReadFileCache (default false). */
|
|
1422
|
+
disableReadFileCache: boolean
|
|
1361
1423
|
/**
|
|
1362
1424
|
* Phase 3 context condensation: the model's real context window in
|
|
1363
1425
|
* tokens, when explicitly configured (undefined = resolve live from
|
|
@@ -1536,6 +1598,17 @@ export class HeadlessSession {
|
|
|
1536
1598
|
* BudgetTracker either way.
|
|
1537
1599
|
*/
|
|
1538
1600
|
private sessionEnded = false
|
|
1601
|
+
/**
|
|
1602
|
+
* Evidence-gated completion (fabrication fix, 2026-09-01): the outcome of
|
|
1603
|
+
* the last attempt_completion claim-verification pass, when
|
|
1604
|
+
* evidenceRequiredCompletion was on and the result contained
|
|
1605
|
+
* machine-checkable claims. Carried onto the SessionResult (see
|
|
1606
|
+
* src/engine/types.ts SessionResult.verification) so callers can
|
|
1607
|
+
* distinguish "success claim independently verified" from "success
|
|
1608
|
+
* accepted on prose alone". undefined when the gate didn't run (flag off,
|
|
1609
|
+
* or no claims to check).
|
|
1610
|
+
*/
|
|
1611
|
+
private lastCompletionVerification: SessionCompletionVerification | undefined = undefined
|
|
1539
1612
|
/**
|
|
1540
1613
|
* True once the real context window has been resolved (from config, the
|
|
1541
1614
|
* OpenRouter models endpoint, or the conservative default) — the lookup
|
|
@@ -1680,12 +1753,14 @@ export class HeadlessSession {
|
|
|
1680
1753
|
requireExplicitCompletion: config.requireExplicitCompletion ?? false,
|
|
1681
1754
|
patchLocalToolSchemas: config.patchLocalToolSchemas ?? false,
|
|
1682
1755
|
verifyBeforeCompletion: config.verifyBeforeCompletion ?? false,
|
|
1756
|
+
evidenceRequiredCompletion: config.evidenceRequiredCompletion ?? false,
|
|
1683
1757
|
trackCost: config.trackCost ?? true,
|
|
1684
1758
|
requireArtifactBeforeCompletion: config.requireArtifactBeforeCompletion ?? false,
|
|
1685
1759
|
requireArtifactPathPattern: config.requireArtifactPathPattern,
|
|
1686
1760
|
requireArtifactMinCitations: config.requireArtifactMinCitations,
|
|
1687
1761
|
requireArtifactSections: config.requireArtifactSections,
|
|
1688
1762
|
guardLargeOverwrites: config.guardLargeOverwrites ?? false,
|
|
1763
|
+
disableReadFileCache: config.disableReadFileCache ?? false,
|
|
1689
1764
|
// Deliberately left undefined when the caller didn't configure it:
|
|
1690
1765
|
// the live OpenRouter models-endpoint lookup in maybeCondenseHistory
|
|
1691
1766
|
// resolves the real context window (never hardcode a stale number —
|
|
@@ -1735,6 +1810,7 @@ export class HeadlessSession {
|
|
|
1735
1810
|
decisionPollIntervalMs: this.config.decisionPollIntervalMs,
|
|
1736
1811
|
permissions: this.config.permissions,
|
|
1737
1812
|
guardLargeOverwrites: this.config.guardLargeOverwrites,
|
|
1813
|
+
disableReadFileCache: this.config.disableReadFileCache,
|
|
1738
1814
|
// Live worker monitoring: mirror ask_followup_question's
|
|
1739
1815
|
// .harness.needs-decision marker lifecycle on the session's
|
|
1740
1816
|
// event feed (decision_blocked / decision_answered). Non-fatal.
|
|
@@ -3242,6 +3318,57 @@ export class HeadlessSession {
|
|
|
3242
3318
|
// write_to_file — see the doc comment where it's set for why this
|
|
3243
3319
|
// exists as a separate flag.
|
|
3244
3320
|
let lastWriteToolFailed = false
|
|
3321
|
+
// 2026-09-02: real, confirmed bug -- lastWriteToolFailed only ever
|
|
3322
|
+
// gets updated when edit_file/write_to_file/set_indentation is
|
|
3323
|
+
// called AGAIN, so once ANY such call fails, this stayed
|
|
3324
|
+
// permanently true for the REST of the session if the model never
|
|
3325
|
+
// touched that tool again -- even when it correctly determined, by
|
|
3326
|
+
// re-reading the real file, that no further edit was needed at
|
|
3327
|
+
// all. Verified live: a session read the file, correctly
|
|
3328
|
+
// concluded the /health handler it was asked to add already
|
|
3329
|
+
// existed and worked, and every subsequent honest, accurate
|
|
3330
|
+
// attempt_completion was deferred anyway with a stale "fix the
|
|
3331
|
+
// issue, make the edit succeed" message that no longer applied --
|
|
3332
|
+
// 15+ consecutive identical deferrals with no bounded-failure
|
|
3333
|
+
// kill-switch to end it (would have spun to the iteration cap).
|
|
3334
|
+
// Tracks the failed call's resolved target path so a genuine
|
|
3335
|
+
// re-verification (a successful read_file of that SAME file)
|
|
3336
|
+
// clears the flag -- narrow and hard to game (it requires actually
|
|
3337
|
+
// re-reading the exact file that failed to edit), unlike a blanket
|
|
3338
|
+
// reset on any successful tool call of any kind.
|
|
3339
|
+
let lastWriteToolFailedTarget: string | undefined
|
|
3340
|
+
// 2026-09-02 (same-day follow-up to the re-read clearing above): the
|
|
3341
|
+
// re-read escape hatch clears lastWriteToolFailed on the assumption
|
|
3342
|
+
// that a model which re-reads the file it failed to edit has
|
|
3343
|
+
// concluded "no edit was needed" and is about to finish honestly.
|
|
3344
|
+
// Verified live (followthrough sweep, "add an entry to a JSON array"
|
|
3345
|
+
// case): a session's edit_file failed, it re-read tasks.json exactly
|
|
3346
|
+
// as the re-read hatch expects, then called attempt_completion
|
|
3347
|
+
// claiming *"I added {\"id\": 7, \"name\": \"lint\"} to the tasks
|
|
3348
|
+
// array in tasks.json."* — an edit it never actually landed. The
|
|
3349
|
+
// re-read had cleared the flag, so the completion sailed through and
|
|
3350
|
+
// the sweep scored it a fabrication. The re-read alone can't tell
|
|
3351
|
+
// "no edit needed" (legit) from "edit still not made" (fabrication);
|
|
3352
|
+
// the completion prose is the discriminator. This flag records that
|
|
3353
|
+
// a failed write was cleared ONLY by a re-read (never by a real
|
|
3354
|
+
// successful write since); at completion time, if the result also
|
|
3355
|
+
// asserts an edit was made, the completion is deferred. A later
|
|
3356
|
+
// genuinely successful write tool call clears it for good.
|
|
3357
|
+
let writeFailedClearedByReReadOnly = false
|
|
3358
|
+
let reReadClearedWriteSummary: string | undefined
|
|
3359
|
+
// Second layer of defense alongside the lastWriteToolFailedTarget
|
|
3360
|
+
// fix above: even a LEGITIMATE reason to keep deferring (a real,
|
|
3361
|
+
// still-unfixed failure) has no bounded-failure kill-switch on this
|
|
3362
|
+
// specific path today — verified live it can spin to the full
|
|
3363
|
+
// iteration cap on unchanging identical deferrals with zero new
|
|
3364
|
+
// information, the same failure shape as issue #26's fix already
|
|
3365
|
+
// handled for the identical-tool-call-streak case but never for
|
|
3366
|
+
// this one. Counts consecutive verifyBeforeCompletion deferrals
|
|
3367
|
+
// carrying the SAME reason string; resets whenever the reason
|
|
3368
|
+
// changes (a changing reason means real progress/new information is
|
|
3369
|
+
// happening) or a real tool call succeeds.
|
|
3370
|
+
let consecutiveDeferralReason: string | undefined
|
|
3371
|
+
let consecutiveDeferralStreak = 0
|
|
3245
3372
|
// See HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
|
|
3246
3373
|
// comment (issue #143). Set true the first time this session calls
|
|
3247
3374
|
// execute_command, write_to_file, or edit_file — regardless of
|
|
@@ -3249,6 +3376,23 @@ export class HeadlessSession {
|
|
|
3249
3376
|
// tried to produce a real effect; lastExecuteCommandFailed/
|
|
3250
3377
|
// lastWriteToolFailed above separately catch a FAILED attempt).
|
|
3251
3378
|
let hasCalledArtifactTool = false
|
|
3379
|
+
// 2026-09-02: verified live in the fabrication sweep's e1000 case —
|
|
3380
|
+
// the requireArtifactBeforeCompletion deferral (below) has no
|
|
3381
|
+
// bounded-failure kill-switch of its own, unlike the
|
|
3382
|
+
// verifyBeforeCompletion deferral (consecutiveDeferralStreak) and the
|
|
3383
|
+
// identical-call streak (issue #26). A session that never once calls
|
|
3384
|
+
// a real artifact tool, keeps re-issuing attempt_completion, and
|
|
3385
|
+
// generates a large inline "report" each turn spun 17+ iterations /
|
|
3386
|
+
// 971s before the duration cap finally killed it — the model even
|
|
3387
|
+
// correctly diagnosed its own fabrication ("I never actually called
|
|
3388
|
+
// write_to_file… the report I gave was the real output I expected to
|
|
3389
|
+
// see AFTER running the gate") and kept going anyway. Count
|
|
3390
|
+
// consecutive hits of that specific deferral; once it reaches
|
|
3391
|
+
// consecutiveErrorLimit, end as a bounded failure. No reset needed:
|
|
3392
|
+
// the branch only fires while hasCalledArtifactTool is false, and the
|
|
3393
|
+
// first real execute_command/write_to_file/edit_file call (even a
|
|
3394
|
+
// failed one) flips that permanently.
|
|
3395
|
+
let artifactBeforeCompletionDeferrals = 0
|
|
3252
3396
|
// Concrete command + truncated error output for the failure above,
|
|
3253
3397
|
// so the attempt_completion rejection below can restate WHAT failed
|
|
3254
3398
|
// instead of pointing at it abstractly — by the time a model reaches
|
|
@@ -3677,9 +3821,30 @@ export class HeadlessSession {
|
|
|
3677
3821
|
!messages.some(
|
|
3678
3822
|
(m) => m.role === "tool" && m.name === "execute_command" && /\d/.test(String(m.content ?? "")),
|
|
3679
3823
|
)
|
|
3824
|
+
// staleReReadEditClaim: a failed write was cleared by a re-read
|
|
3825
|
+
// ONLY (never a real successful write since — see
|
|
3826
|
+
// writeFailedClearedByReReadOnly's doc), yet this completion's prose
|
|
3827
|
+
// still asserts an edit was actually made. The re-read hatch exists
|
|
3828
|
+
// for the honest "no edit was needed" finish; a completion that
|
|
3829
|
+
// claims it *did* edit contradicts the still-unlanded write and is
|
|
3830
|
+
// deferred. Deliberately narrow: only affirmative "I/we added|
|
|
3831
|
+
// wrote|created|…", "added|inserted|… <x> (in)to <file>", or "the
|
|
3832
|
+
// edit/change has been made/applied" shapes — "already exists",
|
|
3833
|
+
// "no edit needed", "was already correct" carry no change verb and
|
|
3834
|
+
// still pass cleanly.
|
|
3835
|
+
const completionAssertsEditMade =
|
|
3836
|
+
/\b(?:I|we|I've|we've|I have|we have)\s+(?:just\s+|now\s+|successfully\s+)?(?:added|inserted|appended|wrote|written|created|updated|edited|modified|applied|replaced|changed)\b/i.test(
|
|
3837
|
+
result,
|
|
3838
|
+
) ||
|
|
3839
|
+
/\b(?:added|inserted|appended|wrote|created|placed)\s+[^.\n]{0,80}?\b(?:in)?to\b\s+\S*[A-Za-z0-9_-]/i.test(result) ||
|
|
3840
|
+
/\bthe\s+(?:edit|change|fix|update|modification|entry|line|function|field)\s+(?:was|has been|is now)\s+(?:made|applied|written|added|inserted|in place|complete)\b/i.test(
|
|
3841
|
+
result,
|
|
3842
|
+
) ||
|
|
3843
|
+
/\bhas been\s+(?:added|inserted|appended|written|updated|applied|modified|replaced)\b/i.test(result)
|
|
3844
|
+
const staleReReadEditClaim = writeFailedClearedByReReadOnly && completionAssertsEditMade
|
|
3680
3845
|
if (
|
|
3681
3846
|
this.config.verifyBeforeCompletion &&
|
|
3682
|
-
(lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim)
|
|
3847
|
+
(lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim || staleReReadEditClaim)
|
|
3683
3848
|
) {
|
|
3684
3849
|
// Verified live 2026-08-29 (joeos issue #26, rounds 17 + 20):
|
|
3685
3850
|
// the `continue` at the end of this block skips the
|
|
@@ -3716,7 +3881,9 @@ export class HeadlessSession {
|
|
|
3716
3881
|
identicalCallStreak = 0
|
|
3717
3882
|
identicalCallNudgeInjected = false
|
|
3718
3883
|
excludedToolCooldowns.delete("execute_command")
|
|
3719
|
-
const reason =
|
|
3884
|
+
const reason = staleReReadEditClaim
|
|
3885
|
+
? "edit claimed but never landed (only a re-read since the failed write)"
|
|
3886
|
+
: unsupportedMeasurementClaim
|
|
3720
3887
|
? "unsupported measurement claim"
|
|
3721
3888
|
: lastExecuteCommandFailed
|
|
3722
3889
|
? "last execute_command failed"
|
|
@@ -3729,7 +3896,9 @@ export class HeadlessSession {
|
|
|
3729
3896
|
role: "tool",
|
|
3730
3897
|
tool_call_id: sibling.id,
|
|
3731
3898
|
name: sibling.name,
|
|
3732
|
-
content:
|
|
3899
|
+
content: staleReReadEditClaim
|
|
3900
|
+
? "[System: not executed — attempt_completion was deferred because your result claims an edit was made, but the edit_file/write_to_file call for it failed and you have only re-read the file since (never landed a successful write); re-issue this call if still needed.]"
|
|
3901
|
+
: unsupportedMeasurementClaim
|
|
3733
3902
|
? "[System: not executed — attempt_completion was deferred because it makes a specific measurement/benchmark claim with no execute_command output in this session's history containing any supporting number; re-issue this call if still needed.]"
|
|
3734
3903
|
: lastExecuteCommandFailed
|
|
3735
3904
|
? "[System: not executed — attempt_completion was deferred because the last command you ran ended in an error; re-issue this call if still needed.]"
|
|
@@ -3740,7 +3909,11 @@ export class HeadlessSession {
|
|
|
3740
3909
|
role: "tool",
|
|
3741
3910
|
tool_call_id: completionCall.id,
|
|
3742
3911
|
name: "attempt_completion",
|
|
3743
|
-
content:
|
|
3912
|
+
content: staleReReadEditClaim
|
|
3913
|
+
? "[System: attempt_completion was NOT accepted. Your result describes an edit as made (e.g. \"added …\", \"the entry has been added\"), but the edit_file/write_to_file call for it failed and the only thing you have done since is re-read the file — no successful write ever landed:\n" +
|
|
3914
|
+
`${reReadClearedWriteSummary ?? "(edit output no longer available)"}\n` +
|
|
3915
|
+
"Either actually make the edit succeed and re-issue attempt_completion, or, if no edit was truly needed, restate the result to say so plainly (e.g. \"no change was required\") without claiming an edit you did not land.]"
|
|
3916
|
+
: unsupportedMeasurementClaim
|
|
3744
3917
|
? "[System: attempt_completion was NOT accepted. Your result claims a specific measurement/benchmark, but no execute_command output anywhere in this session contains a supporting number. Either run the real command that produces this evidence and re-issue attempt_completion, or restate the result without the unsupported claim.]"
|
|
3745
3918
|
: lastExecuteCommandFailed
|
|
3746
3919
|
? "[System: attempt_completion was NOT accepted. The most recent command you ran ended in an error, and you have not run a command since that succeeded:\n" +
|
|
@@ -3751,8 +3924,28 @@ export class HeadlessSession {
|
|
|
3751
3924
|
"Fix the issue, make the edit succeed, and only call attempt_completion again once it actually applied.]",
|
|
3752
3925
|
})
|
|
3753
3926
|
this.logger.warn("[loop] attempt_completion deferred", { iteration, reason })
|
|
3927
|
+
if (reason === consecutiveDeferralReason) {
|
|
3928
|
+
consecutiveDeferralStreak++
|
|
3929
|
+
} else {
|
|
3930
|
+
consecutiveDeferralReason = reason
|
|
3931
|
+
consecutiveDeferralStreak = 1
|
|
3932
|
+
}
|
|
3933
|
+
if (consecutiveDeferralStreak >= this.config.consecutiveErrorLimit) {
|
|
3934
|
+
return this.boundedFailure(
|
|
3935
|
+
"consecutive completion deferrals (same reason, no new evidence)",
|
|
3936
|
+
iteration,
|
|
3937
|
+
toolCalls,
|
|
3938
|
+
consecutiveDeferralStreak,
|
|
3939
|
+
)
|
|
3940
|
+
}
|
|
3754
3941
|
continue
|
|
3755
3942
|
}
|
|
3943
|
+
// A turn that reaches here without hitting the deferral branch
|
|
3944
|
+
// above represents real progress (a normal tool call ran, or
|
|
3945
|
+
// this SPECIFIC completion was actually accepted) — the streak
|
|
3946
|
+
// above only means anything as CONSECUTIVE identical deferrals.
|
|
3947
|
+
consecutiveDeferralReason = undefined
|
|
3948
|
+
consecutiveDeferralStreak = 0
|
|
3756
3949
|
|
|
3757
3950
|
// requireArtifactBeforeCompletion guardrail (issue #143) — see
|
|
3758
3951
|
// HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
|
|
@@ -3782,6 +3975,15 @@ export class HeadlessSession {
|
|
|
3782
3975
|
iteration,
|
|
3783
3976
|
reason: "no artifact-producing tool call in this session",
|
|
3784
3977
|
})
|
|
3978
|
+
artifactBeforeCompletionDeferrals++
|
|
3979
|
+
if (artifactBeforeCompletionDeferrals >= this.config.consecutiveErrorLimit) {
|
|
3980
|
+
return this.boundedFailure(
|
|
3981
|
+
"repeated attempt_completion with no artifact-producing tool call ever made",
|
|
3982
|
+
iteration,
|
|
3983
|
+
toolCalls,
|
|
3984
|
+
artifactBeforeCompletionDeferrals,
|
|
3985
|
+
)
|
|
3986
|
+
}
|
|
3785
3987
|
continue
|
|
3786
3988
|
}
|
|
3787
3989
|
|
|
@@ -3953,9 +4155,139 @@ export class HeadlessSession {
|
|
|
3953
4155
|
continue
|
|
3954
4156
|
}
|
|
3955
4157
|
|
|
4158
|
+
// Evidence-gated completion (fabrication fix, 2026-09-01 — see
|
|
4159
|
+
// src/engine/claims.ts's module doc for the full writeup): when
|
|
4160
|
+
// evidenceRequiredCompletion is on, every machine-checkable claim
|
|
4161
|
+
// in the completion's result text (a file exists, a specific
|
|
4162
|
+
// command passed, serial markers appear, a PR exists) must be
|
|
4163
|
+
// independently verified against ground truth — the filesystem,
|
|
4164
|
+
// a real re-run of the exact command, the newest serial log, and
|
|
4165
|
+
// real git history — BEFORE the completion is accepted. This is
|
|
4166
|
+
// the structural backstop for the FINAL_REPORT's central finding:
|
|
4167
|
+
// a session claimed "all three hard gates pass" with a fabricated
|
|
4168
|
+
// serial-log excerpt when the driver was never merged and the
|
|
4169
|
+
// claimed Makefile target didn't exist. Fail-closed: any
|
|
4170
|
+
// unverifiable claim defers the completion with a corrective
|
|
4171
|
+
// message naming the specific claim, and emits an
|
|
4172
|
+
// `unverified_claim` feed event so downstream consumers (eval,
|
|
4173
|
+
// selfplay miner, orchestrator) can see WHY the completion was
|
|
4174
|
+
// not accepted. A result with NO machine-checkable claims (pure
|
|
4175
|
+
// prose) does not gate — but it also can never *pass* a gate.
|
|
4176
|
+
this.lastCompletionVerification = undefined
|
|
4177
|
+
if (this.config.evidenceRequiredCompletion) {
|
|
4178
|
+
const claims = extractClaims(result)
|
|
4179
|
+
if (claims.length > 0) {
|
|
4180
|
+
const verification = await verifyClaims(claims, {
|
|
4181
|
+
workspaceRoot: this.config.workspaceRoot,
|
|
4182
|
+
permissions: this.executor.permissions,
|
|
4183
|
+
})
|
|
4184
|
+
this.lastCompletionVerification = {
|
|
4185
|
+
claimsChecked: claims.length,
|
|
4186
|
+
claimsPassed: verification.filter((v) => v.verified).length,
|
|
4187
|
+
claimsUnverified: verification.filter((v) => !v.verified).length,
|
|
4188
|
+
}
|
|
4189
|
+
if (!allClaimsVerified(verification)) {
|
|
4190
|
+
// Identical-call guardrail reset — the same live
|
|
4191
|
+
// failure the verifyBeforeCompletion deferral above
|
|
4192
|
+
// documents (joeos issue #26, rounds 17 + 20): the
|
|
4193
|
+
// `continue` at the end of this block skips the
|
|
4194
|
+
// identical-call streak update below (it's part of
|
|
4195
|
+
// the normal per-iteration `calls` processing this
|
|
4196
|
+
// branch exits before reaching), so without this
|
|
4197
|
+
// reset lastCallBatchSignature/lastCallBatchNames/
|
|
4198
|
+
// identicalCallStreak stay FROZEN at whatever they
|
|
4199
|
+
// were when this deferral first started firing —
|
|
4200
|
+
// typically two identical execute_command failures
|
|
4201
|
+
// in a row, which is often exactly what triggers a
|
|
4202
|
+
// fabricated-completion deferral in the first
|
|
4203
|
+
// place (a model re-calling the same failing gate).
|
|
4204
|
+
// Every subsequent deferred-completion turn then
|
|
4205
|
+
// re-enters the request-prep cooldown-refresh loop
|
|
4206
|
+
// with identicalCallGuardActive still true and
|
|
4207
|
+
// lastCallBatchNames still ["execute_command"],
|
|
4208
|
+
// re-arming that tool's exclusion to the full
|
|
4209
|
+
// cooldown value EVERY turn before it ever ticks
|
|
4210
|
+
// down — the model has no legal move
|
|
4211
|
+
// (attempt_completion deferred, execute_command
|
|
4212
|
+
// excluded) and just keeps re-calling
|
|
4213
|
+
// attempt_completion, which is exactly the input
|
|
4214
|
+
// that keeps re-triggering this same `continue`
|
|
4215
|
+
// path. Confirmed live: 30+ iterations spinning
|
|
4216
|
+
// between "attempt_completion deferred" and an
|
|
4217
|
+
// unchanging "excludedToolCooldowns:
|
|
4218
|
+
// {execute_command: 4}" until the iteration cap was
|
|
4219
|
+
// hit. A deferred completion is definitionally not
|
|
4220
|
+
// a repeat of whatever tool-call batch came before
|
|
4221
|
+
// it, so the guard has no reason to stay active
|
|
4222
|
+
// into the next turn.
|
|
4223
|
+
lastCallBatchSignature = null
|
|
4224
|
+
lastCallBatchNames = []
|
|
4225
|
+
identicalCallStreak = 0
|
|
4226
|
+
identicalCallNudgeInjected = false
|
|
4227
|
+
excludedToolCooldowns.delete("execute_command")
|
|
4228
|
+
const firstBad = verification.find((v) => !v.verified)
|
|
4229
|
+
const unverifiedLabels = verification
|
|
4230
|
+
.filter((v) => !v.verified)
|
|
4231
|
+
.map((v) => claimLabel(v.claim))
|
|
4232
|
+
.join(", ")
|
|
4233
|
+
for (const sibling of calls) {
|
|
4234
|
+
if (sibling.id === completionCall.id) {
|
|
4235
|
+
continue
|
|
4236
|
+
}
|
|
4237
|
+
messages.push({
|
|
4238
|
+
role: "tool",
|
|
4239
|
+
tool_call_id: sibling.id,
|
|
4240
|
+
name: sibling.name,
|
|
4241
|
+
content:
|
|
4242
|
+
"[System: not executed — attempt_completion was deferred because your result makes claims that could not be verified against the real workspace; re-issue this call if still needed.]",
|
|
4243
|
+
})
|
|
4244
|
+
}
|
|
4245
|
+
messages.push({
|
|
4246
|
+
role: "tool",
|
|
4247
|
+
tool_call_id: completionCall.id,
|
|
4248
|
+
name: "attempt_completion",
|
|
4249
|
+
content:
|
|
4250
|
+
"[System: attempt_completion was NOT accepted. Your result claims: " +
|
|
4251
|
+
`${unverifiedLabels}. None of these could be independently confirmed: ` +
|
|
4252
|
+
`${firstBad?.detail ?? "no evidence found"}. ` +
|
|
4253
|
+
"Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
|
|
4254
|
+
"Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and re-issue attempt_completion, " +
|
|
4255
|
+
"or restate the result to only claim what you have actually verified.]",
|
|
4256
|
+
})
|
|
4257
|
+
this.logger.warn("[loop] attempt_completion deferred — unverifiable claims in result", {
|
|
4258
|
+
iteration,
|
|
4259
|
+
claims,
|
|
4260
|
+
verification: verification.map((v) => ({ verified: v.verified, detail: v.detail })),
|
|
4261
|
+
})
|
|
4262
|
+
this.scheduleAux(() =>
|
|
4263
|
+
this.emitEvent(
|
|
4264
|
+
"unverified_claim",
|
|
4265
|
+
() =>
|
|
4266
|
+
this.eventFeed.unverifiedClaim({
|
|
4267
|
+
iteration,
|
|
4268
|
+
claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
|
|
4269
|
+
claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
|
|
4270
|
+
claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
|
|
4271
|
+
detail: firstBad?.detail ?? "",
|
|
4272
|
+
}),
|
|
4273
|
+
{ iteration },
|
|
4274
|
+
),
|
|
4275
|
+
)
|
|
4276
|
+
continue
|
|
4277
|
+
}
|
|
4278
|
+
}
|
|
4279
|
+
}
|
|
4280
|
+
|
|
3956
4281
|
this.logger.info("[loop] attempt_completion received — success", { iteration })
|
|
3957
4282
|
const reportPath = await this.persistFinalReport(iteration, result)
|
|
3958
|
-
return {
|
|
4283
|
+
return {
|
|
4284
|
+
status: "success",
|
|
4285
|
+
result,
|
|
4286
|
+
iterations: iteration,
|
|
4287
|
+
toolCalls: toolCalls + 1,
|
|
4288
|
+
reportPath,
|
|
4289
|
+
verification: this.lastCompletionVerification,
|
|
4290
|
+
}
|
|
3959
4291
|
}
|
|
3960
4292
|
|
|
3961
4293
|
// 6b. Text-only reply (no tool_calls) → pragmatic success fallback,
|
|
@@ -4082,9 +4414,93 @@ export class HeadlessSession {
|
|
|
4082
4414
|
artifactRejectionStreak = 0
|
|
4083
4415
|
artifactRejectionNudgeInjected = false
|
|
4084
4416
|
if (text && !this.config.requireExplicitCompletion) {
|
|
4417
|
+
// Evidence-gated completion (fabrication fix, 2026-09-01)
|
|
4418
|
+
// — the text-only success fallback is a REAL bypass for
|
|
4419
|
+
// cloud sessions: requireExplicitCompletion defaults OFF
|
|
4420
|
+
// for the cloud backend, so with --require-evidence a
|
|
4421
|
+
// cloud model could dump prose ("all three hard gates
|
|
4422
|
+
// pass…") and be recorded as success with ZERO
|
|
4423
|
+
// evidence, exactly the fabrication shape the gate
|
|
4424
|
+
// exists to stop. When evidenceRequiredCompletion is
|
|
4425
|
+
// ON, a text-only reply is treated as a completion
|
|
4426
|
+
// CANDIDATE and runs the SAME extract/verify gate as an
|
|
4427
|
+
// attempt_completion: pure prose (no machine-checkable
|
|
4428
|
+
// claims) or fully-verified claims are accepted; any
|
|
4429
|
+
// unverifiable claim defers with the corrective nudge
|
|
4430
|
+
// and an unverified_claim event, never a success.
|
|
4431
|
+
this.lastCompletionVerification = undefined
|
|
4432
|
+
if (this.config.evidenceRequiredCompletion) {
|
|
4433
|
+
const claims = extractClaims(text)
|
|
4434
|
+
if (claims.length > 0) {
|
|
4435
|
+
const verification = await verifyClaims(claims, {
|
|
4436
|
+
workspaceRoot: this.config.workspaceRoot,
|
|
4437
|
+
permissions: this.executor.permissions,
|
|
4438
|
+
})
|
|
4439
|
+
this.lastCompletionVerification = {
|
|
4440
|
+
claimsChecked: claims.length,
|
|
4441
|
+
claimsPassed: verification.filter((v) => v.verified).length,
|
|
4442
|
+
claimsUnverified: verification.filter((v) => !v.verified).length,
|
|
4443
|
+
}
|
|
4444
|
+
if (!allClaimsVerified(verification)) {
|
|
4445
|
+
// Same identical-call guardrail reset as the
|
|
4446
|
+
// attempt_completion deferral above — a text-only
|
|
4447
|
+
// reply is definitionally not a repeat of the
|
|
4448
|
+
// previous tool-call batch.
|
|
4449
|
+
lastCallBatchSignature = null
|
|
4450
|
+
lastCallBatchNames = []
|
|
4451
|
+
identicalCallStreak = 0
|
|
4452
|
+
identicalCallNudgeInjected = false
|
|
4453
|
+
excludedToolCooldowns.delete("execute_command")
|
|
4454
|
+
const firstBad = verification.find((v) => !v.verified)
|
|
4455
|
+
const unverifiedLabels = verification
|
|
4456
|
+
.filter((v) => !v.verified)
|
|
4457
|
+
.map((v) => claimLabel(v.claim))
|
|
4458
|
+
.join(", ")
|
|
4459
|
+
messages.push({
|
|
4460
|
+
role: "user",
|
|
4461
|
+
content:
|
|
4462
|
+
`[System: your text-only reply was NOT accepted as a completion. It claims: ${unverifiedLabels}. ` +
|
|
4463
|
+
`None of these could be independently confirmed: ${firstBad?.detail ?? "no evidence found"}. ` +
|
|
4464
|
+
"Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
|
|
4465
|
+
"Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and then call attempt_completion, " +
|
|
4466
|
+
"or restate your answer to only claim what you have actually verified.]",
|
|
4467
|
+
})
|
|
4468
|
+
this.logger.warn("[loop] text-only reply NOT accepted — unverifiable claims", {
|
|
4469
|
+
iteration,
|
|
4470
|
+
claims,
|
|
4471
|
+
verification: verification.map((v) => ({
|
|
4472
|
+
verified: v.verified,
|
|
4473
|
+
detail: v.detail,
|
|
4474
|
+
})),
|
|
4475
|
+
})
|
|
4476
|
+
this.scheduleAux(() =>
|
|
4477
|
+
this.emitEvent(
|
|
4478
|
+
"unverified_claim",
|
|
4479
|
+
() =>
|
|
4480
|
+
this.eventFeed.unverifiedClaim({
|
|
4481
|
+
iteration,
|
|
4482
|
+
claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
|
|
4483
|
+
claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
|
|
4484
|
+
claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
|
|
4485
|
+
detail: firstBad?.detail ?? "",
|
|
4486
|
+
}),
|
|
4487
|
+
{ iteration },
|
|
4488
|
+
),
|
|
4489
|
+
)
|
|
4490
|
+
continue
|
|
4491
|
+
}
|
|
4492
|
+
}
|
|
4493
|
+
}
|
|
4085
4494
|
this.logger.info("[loop] text-only reply (no tool calls) — success", { iteration })
|
|
4086
4495
|
const reportPath = await this.persistFinalReport(iteration, text)
|
|
4087
|
-
return {
|
|
4496
|
+
return {
|
|
4497
|
+
status: "success",
|
|
4498
|
+
result: text,
|
|
4499
|
+
iterations: iteration,
|
|
4500
|
+
toolCalls,
|
|
4501
|
+
reportPath,
|
|
4502
|
+
...(this.lastCompletionVerification ? { verification: this.lastCompletionVerification } : {}),
|
|
4503
|
+
}
|
|
4088
4504
|
}
|
|
4089
4505
|
// Empty reply, or a text reply that requireExplicitCompletion
|
|
4090
4506
|
// refuses to treat as final: nudge and count as a mistake.
|
|
@@ -4296,8 +4712,35 @@ export class HeadlessSession {
|
|
|
4296
4712
|
? resultContent.slice(0, 500)
|
|
4297
4713
|
: JSON.stringify(resultContent).slice(0, 500)
|
|
4298
4714
|
lastWriteToolSummary = `${call.name} ${target}\n${output}`
|
|
4715
|
+
lastWriteToolFailedTarget = toolCallPathArg(this.config.workspaceRoot, call)
|
|
4299
4716
|
} else {
|
|
4300
4717
|
lastWriteToolSummary = undefined
|
|
4718
|
+
lastWriteToolFailedTarget = undefined
|
|
4719
|
+
// A genuinely successful write resolves the situation for
|
|
4720
|
+
// real — the re-read-only clearing no longer needs to gate
|
|
4721
|
+
// anything (see writeFailedClearedByReReadOnly's doc above).
|
|
4722
|
+
writeFailedClearedByReReadOnly = false
|
|
4723
|
+
reReadClearedWriteSummary = undefined
|
|
4724
|
+
}
|
|
4725
|
+
}
|
|
4726
|
+
// See lastWriteToolFailedTarget's doc comment above: a
|
|
4727
|
+
// successful read_file of the EXACT file a write tool just
|
|
4728
|
+
// failed to edit is real re-verification evidence, not just
|
|
4729
|
+
// time passing — clear the stale-failure flag so a
|
|
4730
|
+
// subsequent honest completion (including "no edit was
|
|
4731
|
+
// actually needed") isn't blocked by a failure the model
|
|
4732
|
+
// has since genuinely re-checked.
|
|
4733
|
+
if (call.name === "read_file" && !isError && lastWriteToolFailed) {
|
|
4734
|
+
const readTarget = toolCallPathArg(this.config.workspaceRoot, call)
|
|
4735
|
+
if (readTarget !== undefined && readTarget === lastWriteToolFailedTarget) {
|
|
4736
|
+
lastWriteToolFailed = false
|
|
4737
|
+
reReadClearedWriteSummary = lastWriteToolSummary
|
|
4738
|
+
lastWriteToolSummary = undefined
|
|
4739
|
+
lastWriteToolFailedTarget = undefined
|
|
4740
|
+
// Remember this was cleared by a re-read ONLY, not by a
|
|
4741
|
+
// real successful write — completion still has to prove
|
|
4742
|
+
// it isn't claiming an edit it never landed.
|
|
4743
|
+
writeFailedClearedByReReadOnly = true
|
|
4301
4744
|
}
|
|
4302
4745
|
}
|
|
4303
4746
|
const targetPath = editToolTargetPath(this.config.workspaceRoot, call)
|
|
@@ -4802,6 +5245,7 @@ export class HeadlessSession {
|
|
|
4802
5245
|
patchLocalToolSchemas: this.config.patchLocalToolSchemas,
|
|
4803
5246
|
verifyBeforeCompletion: this.config.verifyBeforeCompletion,
|
|
4804
5247
|
guardLargeOverwrites: this.config.guardLargeOverwrites,
|
|
5248
|
+
disableReadFileCache: this.config.disableReadFileCache,
|
|
4805
5249
|
llmClient: this.llmClient,
|
|
4806
5250
|
logger: this.logger,
|
|
4807
5251
|
memory: this.config.memory,
|
|
@@ -5311,9 +5755,20 @@ export function summarizeToolArg(name: string, args: Record<string, unknown>): s
|
|
|
5311
5755
|
case "list_files":
|
|
5312
5756
|
case "write_to_file":
|
|
5313
5757
|
case "apply_diff":
|
|
5758
|
+
return str(args.path)
|
|
5314
5759
|
case "search_replace":
|
|
5315
5760
|
case "edit_file":
|
|
5316
|
-
|
|
5761
|
+
// 2026-09-02: real, confirmed bug -- these two tools' native
|
|
5762
|
+
// schemas (src/vendor/zoo-code/.../native-tools/edit_file.ts,
|
|
5763
|
+
// search_replace.ts) declare `file_path`, not `path` (unlike
|
|
5764
|
+
// write_to_file/apply_diff/list_files, which really do use
|
|
5765
|
+
// `path`) -- so this always returned undefined -> "" here,
|
|
5766
|
+
// making the tool_call event's path summary silently blank for
|
|
5767
|
+
// every edit_file/search_replace call. That's exactly what made
|
|
5768
|
+
// live log-watching during a real session unable to show which
|
|
5769
|
+
// file was being edited. Fall back to `path` too in case an
|
|
5770
|
+
// older/alias caller still sends that key.
|
|
5771
|
+
return str(args.file_path) ?? str(args.path)
|
|
5317
5772
|
case "execute_command":
|
|
5318
5773
|
return str(args.command)?.slice(0, 200)
|
|
5319
5774
|
case "update_todo_list":
|
package/src/engine/prompt.ts
CHANGED
|
@@ -660,7 +660,11 @@ export function buildHeadlessConventions(mode: string, customModes: ModeConfig[]
|
|
|
660
660
|
- Commit before finishing: real, working changes must be committed via \`git add\` + \`git commit\` BEFORE calling attempt_completion. Use one commit per logical change with a descriptive message, matching this repo's normal style (run \`git log --oneline\` for examples). If you genuinely have no changes to commit (e.g. a read-only investigation), finish without committing.
|
|
661
661
|
- Test selection during iterative work: after editing a file, run the SPECIFIC test file(s) for what you changed (the \`run_tests\` tool, or \`npx tsx <test-file>\` directly) instead of the full suite on every edit. The full \`npm test\` is still required once, right before attempt_completion, for real confidence.
|
|
662
662
|
- Batch same-file diffs: when several changes target the SAME file, include them as separate SEARCH/REPLACE blocks in ONE apply_diff call — and after any successful edit, re-read the file before composing further diffs, because its content has changed.
|
|
663
|
-
- Waiting on an external check (CI run, registry, live service) is legitimate verification work, but every wait call still counts against your iteration budget: prefer ONE long-running command with an explicit \`timeout\` — e.g. \`gh run watch <id> --interval 30 --exit-status\` — over repeated \`sleep N && gh run list\` polls. (Issue #119: polling burns iterations with no token spend.)
|
|
663
|
+
- Waiting on an external check (CI run, registry, live service) is legitimate verification work, but every wait call still counts against your iteration budget: prefer ONE long-running command with an explicit \`timeout\` — e.g. \`gh run watch <id> --interval 30 --exit-status\` — over repeated \`sleep N && gh run list\` polls. (Issue #119: polling burns iterations with no token spend.)
|
|
664
|
+
- Check the real workspace before assuming a path exists: when a task resembles something you've worked on before, use \`list_files\`/\`read_file\` on THIS workspace first rather than guessing a path from a similar prior task — a plausible-looking guess against the wrong repo/scratch layout still fails, and repeating the same wrong guess wastes the iteration budget without new information.
|
|
665
|
+
- If the same command fails twice in a row, change what you do next — read the real error, try something different, or (if there's genuinely no path forward) report the honest state via \`attempt_completion\` rather than retrying the identical call or repeating the same diagnosis in prose without acting on it. A real, honest "this doesn't work and here's why" is always a valid, complete answer; a stalled loop is not.
|
|
666
|
+
- When \`attempt_completion\` is deferred (rejected because a check hasn't actually passed yet, an edit hasn't succeeded, or a claim isn't backed by real evidence), your VERY NEXT action must be the real fix the deferral message describes — a new command, a corrected edit, or gathering the missing evidence — never another \`attempt_completion\` call on its own. Retrying completion without first taking that corrective action just repeats the same rejection and burns the iteration budget with no new information; it will keep being deferred for the identical reason until you actually address it.
|
|
667
|
+
- Starting a long-running process (a dev server, a background listener) and then checking it — do this as TWO separate \`execute_command\` calls, not one. First call: run the command DIRECTLY, with no trailing \`&\` and no \`bash -c '... &'\` wrapper — e.g. \`python3 server.py\` with \`timeout: 2\`. The tool's own contract already keeps it running in the background past the timeout (you get the output captured so far, not an error); it does NOT need you to background it yourself. Second, separate call: the actual check (\`curl ...\`, \`nc -z ...\`). Manually backgrounding with \`&\` inside a \`bash -c\` (or any subshell) is a real, observed failure mode — the process gets backgrounded INSIDE that subshell, which then exits once the \`&\` returns, and the process can be reaped along with it before your next command in the same chain ever reaches it. If a check right after starting something fails, don't keep editing the file you just started — first suspect the START command's shape.`
|
|
664
668
|
}
|
|
665
669
|
|
|
666
670
|
// ─── Tool selection ──────────────────────────────────────────────────────────
|