headlesscode 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -62,6 +62,14 @@ import type { MemoryStore, RecallResult } from "../memory/types.js"
62
62
  import { createCheckpointService, type CheckpointService } from "../checkpoints/service.js"
63
63
  import { recordSessionUsage, removeLiveUsage, writeLiveUsage } from "./usage.js"
64
64
  import { EventFeed, EVENT_TRUNCATE_CHARS, truncateField } from "./events.js"
65
+ import {
66
+ allClaimsVerified,
67
+ claimLabel,
68
+ extractClaims,
69
+ firstUnverifiedDetail,
70
+ verifyClaims,
71
+ type ClaimVerification,
72
+ } from "./claims.js"
65
73
  import { writeSessionReport } from "./reports.js"
66
74
  import { Logger } from "./logger.js"
67
75
  import { extractEmbeddedToolCall, parseToolCalls } from "./parser.js"
@@ -89,6 +97,7 @@ import type {
89
97
  ParsedToolCall,
90
98
  SessionResult,
91
99
  SessionBudgetUsage,
100
+ SessionCompletionVerification,
92
101
  ToolContext,
93
102
  ToolResult,
94
103
  } from "./types.js"
@@ -1068,6 +1077,37 @@ export interface HeadlessSessionConfig {
1068
1077
  * code-mode backend turns this on.
1069
1078
  */
1070
1079
  verifyBeforeCompletion?: boolean
1080
+ /**
1081
+ * Evidence-gated completion — the fabrication fix (2026-09-01, see
1082
+ * src/engine/claims.ts). When true, `attempt_completion` is refused
1083
+ * unless every machine-checkable claim its result text makes (a file
1084
+ * exists, a specific command passed, serial markers appear, a PR
1085
+ * exists) is independently verified against ground truth:
1086
+ *
1087
+ * - file claim → fs.stat on the resolved workspace path
1088
+ * - command claim → RE-RUN the exact command (same permission gate as
1089
+ * execute_command), require exit 0
1090
+ * - serial marker → grep the newest build/serial-*.log for the claimed
1091
+ * ordered markers
1092
+ * - PR claim → require real git-history evidence of the number
1093
+ *
1094
+ * Ground truth comes from the filesystem and real re-runs — never from
1095
+ * the model's own prose. This is the structural backstop for the exact
1096
+ * failure the FINAL_REPORT documented (§4): a session claimed "all
1097
+ * three hard gates pass" with a fabricated serial-log excerpt when the
1098
+ * driver was never merged and the claimed Makefile target didn't exist.
1099
+ * Fail-closed: any unverifiable claim defers the completion with a
1100
+ * corrective message naming the specific unverified claim.
1101
+ *
1102
+ * Deliberately separate from verifyBeforeCompletion (which is
1103
+ * token/state-based and only looks at the LAST command): this checks
1104
+ * the CONTENT of the completion's claims against the real world,
1105
+ * independent of what the session did or didn't run before.
1106
+ *
1107
+ * Default false; opt-in via --require-evidence or the local code
1108
+ * backend (see cli.ts).
1109
+ */
1110
+ evidenceRequiredCompletion?: boolean
1071
1111
  /**
1072
1112
  * When false, cost is never tracked/accumulated for this session (see
1073
1113
  * BudgetTrackerOptions.trackCost) — local-backend sessions have no real
@@ -1159,6 +1199,24 @@ export interface HeadlessSessionConfig {
1159
1199
  * the local Ollama code-mode backend turns this on.
1160
1200
  */
1161
1201
  guardLargeOverwrites?: boolean
1202
+ /**
1203
+ * read_file/list_files carry a session-scoped "[cache] unchanged, reuse
1204
+ * the earlier result" short-circuit (src/tools/executor.ts) that saves
1205
+ * real, measured token cost against a remote model's per-token bill.
1206
+ * Verified live 2026-09-02 against the local backend: an edit_file
1207
+ * failure told a session to re-read and retry; it DID call read_file
1208
+ * again exactly as instructed, got the cache-hit notice instead of real
1209
+ * content (correct per the mechanism's own design — the safety valve
1210
+ * is "a SECOND consecutive identical call serves real content again"),
1211
+ * never made that second call, and fabricated an attempt_completion
1212
+ * instead. Re-serving a few hundred lines of file content costs a
1213
+ * local session near-nothing (prefill, not generation, against a GPU
1214
+ * with no per-token price) — cheap insurance locally against a much
1215
+ * more expensive failure mode. Default false; the local Ollama code-mode
1216
+ * backend turns this on (same opt-in shape as guardLargeOverwrites
1217
+ * above).
1218
+ */
1219
+ disableReadFileCache?: boolean
1162
1220
  /**
1163
1221
  * Phase 3 context condensation: the model's real context window in
1164
1222
  * tokens, used to decide WHEN to condense (the last request's real
@@ -1346,6 +1404,8 @@ export interface ResolvedSessionConfig {
1346
1404
  patchLocalToolSchemas: boolean
1347
1405
  /** See HeadlessSessionConfig.verifyBeforeCompletion (default false). */
1348
1406
  verifyBeforeCompletion: boolean
1407
+ /** See HeadlessSessionConfig.evidenceRequiredCompletion (default false). */
1408
+ evidenceRequiredCompletion: boolean
1349
1409
  /** See HeadlessSessionConfig.trackCost (default true). */
1350
1410
  trackCost: boolean
1351
1411
  /** See HeadlessSessionConfig.requireArtifactBeforeCompletion (default false). */
@@ -1358,6 +1418,8 @@ export interface ResolvedSessionConfig {
1358
1418
  requireArtifactSections?: string[]
1359
1419
  /** See HeadlessSessionConfig.guardLargeOverwrites (default false). */
1360
1420
  guardLargeOverwrites: boolean
1421
+ /** See HeadlessSessionConfig.disableReadFileCache (default false). */
1422
+ disableReadFileCache: boolean
1361
1423
  /**
1362
1424
  * Phase 3 context condensation: the model's real context window in
1363
1425
  * tokens, when explicitly configured (undefined = resolve live from
@@ -1536,6 +1598,17 @@ export class HeadlessSession {
1536
1598
  * BudgetTracker either way.
1537
1599
  */
1538
1600
  private sessionEnded = false
1601
+ /**
1602
+ * Evidence-gated completion (fabrication fix, 2026-09-01): the outcome of
1603
+ * the last attempt_completion claim-verification pass, when
1604
+ * evidenceRequiredCompletion was on and the result contained
1605
+ * machine-checkable claims. Carried onto the SessionResult (see
1606
+ * src/engine/types.ts SessionResult.verification) so callers can
1607
+ * distinguish "success claim independently verified" from "success
1608
+ * accepted on prose alone". undefined when the gate didn't run (flag off,
1609
+ * or no claims to check).
1610
+ */
1611
+ private lastCompletionVerification: SessionCompletionVerification | undefined = undefined
1539
1612
  /**
1540
1613
  * True once the real context window has been resolved (from config, the
1541
1614
  * OpenRouter models endpoint, or the conservative default) — the lookup
@@ -1680,12 +1753,14 @@ export class HeadlessSession {
1680
1753
  requireExplicitCompletion: config.requireExplicitCompletion ?? false,
1681
1754
  patchLocalToolSchemas: config.patchLocalToolSchemas ?? false,
1682
1755
  verifyBeforeCompletion: config.verifyBeforeCompletion ?? false,
1756
+ evidenceRequiredCompletion: config.evidenceRequiredCompletion ?? false,
1683
1757
  trackCost: config.trackCost ?? true,
1684
1758
  requireArtifactBeforeCompletion: config.requireArtifactBeforeCompletion ?? false,
1685
1759
  requireArtifactPathPattern: config.requireArtifactPathPattern,
1686
1760
  requireArtifactMinCitations: config.requireArtifactMinCitations,
1687
1761
  requireArtifactSections: config.requireArtifactSections,
1688
1762
  guardLargeOverwrites: config.guardLargeOverwrites ?? false,
1763
+ disableReadFileCache: config.disableReadFileCache ?? false,
1689
1764
  // Deliberately left undefined when the caller didn't configure it:
1690
1765
  // the live OpenRouter models-endpoint lookup in maybeCondenseHistory
1691
1766
  // resolves the real context window (never hardcode a stale number —
@@ -1735,6 +1810,7 @@ export class HeadlessSession {
1735
1810
  decisionPollIntervalMs: this.config.decisionPollIntervalMs,
1736
1811
  permissions: this.config.permissions,
1737
1812
  guardLargeOverwrites: this.config.guardLargeOverwrites,
1813
+ disableReadFileCache: this.config.disableReadFileCache,
1738
1814
  // Live worker monitoring: mirror ask_followup_question's
1739
1815
  // .harness.needs-decision marker lifecycle on the session's
1740
1816
  // event feed (decision_blocked / decision_answered). Non-fatal.
@@ -3242,6 +3318,57 @@ export class HeadlessSession {
3242
3318
  // write_to_file — see the doc comment where it's set for why this
3243
3319
  // exists as a separate flag.
3244
3320
  let lastWriteToolFailed = false
3321
+ // 2026-09-02: real, confirmed bug -- lastWriteToolFailed only ever
3322
+ // gets updated when edit_file/write_to_file/set_indentation is
3323
+ // called AGAIN, so once ANY such call fails, this stayed
3324
+ // permanently true for the REST of the session if the model never
3325
+ // touched that tool again -- even when it correctly determined, by
3326
+ // re-reading the real file, that no further edit was needed at
3327
+ // all. Verified live: a session read the file, correctly
3328
+ // concluded the /health handler it was asked to add already
3329
+ // existed and worked, and every subsequent honest, accurate
3330
+ // attempt_completion was deferred anyway with a stale "fix the
3331
+ // issue, make the edit succeed" message that no longer applied --
3332
+ // 15+ consecutive identical deferrals with no bounded-failure
3333
+ // kill-switch to end it (would have spun to the iteration cap).
3334
+ // Tracks the failed call's resolved target path so a genuine
3335
+ // re-verification (a successful read_file of that SAME file)
3336
+ // clears the flag -- narrow and hard to game (it requires actually
3337
+ // re-reading the exact file that failed to edit), unlike a blanket
3338
+ // reset on any successful tool call of any kind.
3339
+ let lastWriteToolFailedTarget: string | undefined
3340
+ // 2026-09-02 (same-day follow-up to the re-read clearing above): the
3341
+ // re-read escape hatch clears lastWriteToolFailed on the assumption
3342
+ // that a model which re-reads the file it failed to edit has
3343
+ // concluded "no edit was needed" and is about to finish honestly.
3344
+ // Verified live (followthrough sweep, "add an entry to a JSON array"
3345
+ // case): a session's edit_file failed, it re-read tasks.json exactly
3346
+ // as the re-read hatch expects, then called attempt_completion
3347
+ // claiming *"I added {\"id\": 7, \"name\": \"lint\"} to the tasks
3348
+ // array in tasks.json."* — an edit it never actually landed. The
3349
+ // re-read had cleared the flag, so the completion sailed through and
3350
+ // the sweep scored it a fabrication. The re-read alone can't tell
3351
+ // "no edit needed" (legit) from "edit still not made" (fabrication);
3352
+ // the completion prose is the discriminator. This flag records that
3353
+ // a failed write was cleared ONLY by a re-read (never by a real
3354
+ // successful write since); at completion time, if the result also
3355
+ // asserts an edit was made, the completion is deferred. A later
3356
+ // genuinely successful write tool call clears it for good.
3357
+ let writeFailedClearedByReReadOnly = false
3358
+ let reReadClearedWriteSummary: string | undefined
3359
+ // Second layer of defense alongside the lastWriteToolFailedTarget
3360
+ // fix above: even a LEGITIMATE reason to keep deferring (a real,
3361
+ // still-unfixed failure) has no bounded-failure kill-switch on this
3362
+ // specific path today — verified live it can spin to the full
3363
+ // iteration cap on unchanging identical deferrals with zero new
3364
+ // information, the same failure shape as issue #26's fix already
3365
+ // handled for the identical-tool-call-streak case but never for
3366
+ // this one. Counts consecutive verifyBeforeCompletion deferrals
3367
+ // carrying the SAME reason string; resets whenever the reason
3368
+ // changes (a changing reason means real progress/new information is
3369
+ // happening) or a real tool call succeeds.
3370
+ let consecutiveDeferralReason: string | undefined
3371
+ let consecutiveDeferralStreak = 0
3245
3372
  // See HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
3246
3373
  // comment (issue #143). Set true the first time this session calls
3247
3374
  // execute_command, write_to_file, or edit_file — regardless of
@@ -3249,6 +3376,23 @@ export class HeadlessSession {
3249
3376
  // tried to produce a real effect; lastExecuteCommandFailed/
3250
3377
  // lastWriteToolFailed above separately catch a FAILED attempt).
3251
3378
  let hasCalledArtifactTool = false
3379
+ // 2026-09-02: verified live in the fabrication sweep's e1000 case —
3380
+ // the requireArtifactBeforeCompletion deferral (below) has no
3381
+ // bounded-failure kill-switch of its own, unlike the
3382
+ // verifyBeforeCompletion deferral (consecutiveDeferralStreak) and the
3383
+ // identical-call streak (issue #26). A session that never once calls
3384
+ // a real artifact tool, keeps re-issuing attempt_completion, and
3385
+ // generates a large inline "report" each turn spun 17+ iterations /
3386
+ // 971s before the duration cap finally killed it — the model even
3387
+ // correctly diagnosed its own fabrication ("I never actually called
3388
+ // write_to_file… the report I gave was the real output I expected to
3389
+ // see AFTER running the gate") and kept going anyway. Count
3390
+ // consecutive hits of that specific deferral; once it reaches
3391
+ // consecutiveErrorLimit, end as a bounded failure. No reset needed:
3392
+ // the branch only fires while hasCalledArtifactTool is false, and the
3393
+ // first real execute_command/write_to_file/edit_file call (even a
3394
+ // failed one) flips that permanently.
3395
+ let artifactBeforeCompletionDeferrals = 0
3252
3396
  // Concrete command + truncated error output for the failure above,
3253
3397
  // so the attempt_completion rejection below can restate WHAT failed
3254
3398
  // instead of pointing at it abstractly — by the time a model reaches
@@ -3677,9 +3821,30 @@ export class HeadlessSession {
3677
3821
  !messages.some(
3678
3822
  (m) => m.role === "tool" && m.name === "execute_command" && /\d/.test(String(m.content ?? "")),
3679
3823
  )
3824
+ // staleReReadEditClaim: a failed write was cleared by a re-read
3825
+ // ONLY (never a real successful write since — see
3826
+ // writeFailedClearedByReReadOnly's doc), yet this completion's prose
3827
+ // still asserts an edit was actually made. The re-read hatch exists
3828
+ // for the honest "no edit was needed" finish; a completion that
3829
+ // claims it *did* edit contradicts the still-unlanded write and is
3830
+ // deferred. Deliberately narrow: only affirmative "I/we added|
3831
+ // wrote|created|…", "added|inserted|… <x> (in)to <file>", or "the
3832
+ // edit/change has been made/applied" shapes — "already exists",
3833
+ // "no edit needed", "was already correct" carry no change verb and
3834
+ // still pass cleanly.
3835
+ const completionAssertsEditMade =
3836
+ /\b(?:I|we|I've|we've|I have|we have)\s+(?:just\s+|now\s+|successfully\s+)?(?:added|inserted|appended|wrote|written|created|updated|edited|modified|applied|replaced|changed)\b/i.test(
3837
+ result,
3838
+ ) ||
3839
+ /\b(?:added|inserted|appended|wrote|created|placed)\s+[^.\n]{0,80}?\b(?:in)?to\b\s+\S*[A-Za-z0-9_-]/i.test(result) ||
3840
+ /\bthe\s+(?:edit|change|fix|update|modification|entry|line|function|field)\s+(?:was|has been|is now)\s+(?:made|applied|written|added|inserted|in place|complete)\b/i.test(
3841
+ result,
3842
+ ) ||
3843
+ /\bhas been\s+(?:added|inserted|appended|written|updated|applied|modified|replaced)\b/i.test(result)
3844
+ const staleReReadEditClaim = writeFailedClearedByReReadOnly && completionAssertsEditMade
3680
3845
  if (
3681
3846
  this.config.verifyBeforeCompletion &&
3682
- (lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim)
3847
+ (lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim || staleReReadEditClaim)
3683
3848
  ) {
3684
3849
  // Verified live 2026-08-29 (joeos issue #26, rounds 17 + 20):
3685
3850
  // the `continue` at the end of this block skips the
@@ -3716,7 +3881,9 @@ export class HeadlessSession {
3716
3881
  identicalCallStreak = 0
3717
3882
  identicalCallNudgeInjected = false
3718
3883
  excludedToolCooldowns.delete("execute_command")
3719
- const reason = unsupportedMeasurementClaim
3884
+ const reason = staleReReadEditClaim
3885
+ ? "edit claimed but never landed (only a re-read since the failed write)"
3886
+ : unsupportedMeasurementClaim
3720
3887
  ? "unsupported measurement claim"
3721
3888
  : lastExecuteCommandFailed
3722
3889
  ? "last execute_command failed"
@@ -3729,7 +3896,9 @@ export class HeadlessSession {
3729
3896
  role: "tool",
3730
3897
  tool_call_id: sibling.id,
3731
3898
  name: sibling.name,
3732
- content: unsupportedMeasurementClaim
3899
+ content: staleReReadEditClaim
3900
+ ? "[System: not executed — attempt_completion was deferred because your result claims an edit was made, but the edit_file/write_to_file call for it failed and you have only re-read the file since (never landed a successful write); re-issue this call if still needed.]"
3901
+ : unsupportedMeasurementClaim
3733
3902
  ? "[System: not executed — attempt_completion was deferred because it makes a specific measurement/benchmark claim with no execute_command output in this session's history containing any supporting number; re-issue this call if still needed.]"
3734
3903
  : lastExecuteCommandFailed
3735
3904
  ? "[System: not executed — attempt_completion was deferred because the last command you ran ended in an error; re-issue this call if still needed.]"
@@ -3740,7 +3909,11 @@ export class HeadlessSession {
3740
3909
  role: "tool",
3741
3910
  tool_call_id: completionCall.id,
3742
3911
  name: "attempt_completion",
3743
- content: unsupportedMeasurementClaim
3912
+ content: staleReReadEditClaim
3913
+ ? "[System: attempt_completion was NOT accepted. Your result describes an edit as made (e.g. \"added …\", \"the entry has been added\"), but the edit_file/write_to_file call for it failed and the only thing you have done since is re-read the file — no successful write ever landed:\n" +
3914
+ `${reReadClearedWriteSummary ?? "(edit output no longer available)"}\n` +
3915
+ "Either actually make the edit succeed and re-issue attempt_completion, or, if no edit was truly needed, restate the result to say so plainly (e.g. \"no change was required\") without claiming an edit you did not land.]"
3916
+ : unsupportedMeasurementClaim
3744
3917
  ? "[System: attempt_completion was NOT accepted. Your result claims a specific measurement/benchmark, but no execute_command output anywhere in this session contains a supporting number. Either run the real command that produces this evidence and re-issue attempt_completion, or restate the result without the unsupported claim.]"
3745
3918
  : lastExecuteCommandFailed
3746
3919
  ? "[System: attempt_completion was NOT accepted. The most recent command you ran ended in an error, and you have not run a command since that succeeded:\n" +
@@ -3751,8 +3924,28 @@ export class HeadlessSession {
3751
3924
  "Fix the issue, make the edit succeed, and only call attempt_completion again once it actually applied.]",
3752
3925
  })
3753
3926
  this.logger.warn("[loop] attempt_completion deferred", { iteration, reason })
3927
+ if (reason === consecutiveDeferralReason) {
3928
+ consecutiveDeferralStreak++
3929
+ } else {
3930
+ consecutiveDeferralReason = reason
3931
+ consecutiveDeferralStreak = 1
3932
+ }
3933
+ if (consecutiveDeferralStreak >= this.config.consecutiveErrorLimit) {
3934
+ return this.boundedFailure(
3935
+ "consecutive completion deferrals (same reason, no new evidence)",
3936
+ iteration,
3937
+ toolCalls,
3938
+ consecutiveDeferralStreak,
3939
+ )
3940
+ }
3754
3941
  continue
3755
3942
  }
3943
+ // A turn that reaches here without hitting the deferral branch
3944
+ // above represents real progress (a normal tool call ran, or
3945
+ // this SPECIFIC completion was actually accepted) — the streak
3946
+ // above only means anything as CONSECUTIVE identical deferrals.
3947
+ consecutiveDeferralReason = undefined
3948
+ consecutiveDeferralStreak = 0
3756
3949
 
3757
3950
  // requireArtifactBeforeCompletion guardrail (issue #143) — see
3758
3951
  // HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
@@ -3782,6 +3975,15 @@ export class HeadlessSession {
3782
3975
  iteration,
3783
3976
  reason: "no artifact-producing tool call in this session",
3784
3977
  })
3978
+ artifactBeforeCompletionDeferrals++
3979
+ if (artifactBeforeCompletionDeferrals >= this.config.consecutiveErrorLimit) {
3980
+ return this.boundedFailure(
3981
+ "repeated attempt_completion with no artifact-producing tool call ever made",
3982
+ iteration,
3983
+ toolCalls,
3984
+ artifactBeforeCompletionDeferrals,
3985
+ )
3986
+ }
3785
3987
  continue
3786
3988
  }
3787
3989
 
@@ -3953,9 +4155,139 @@ export class HeadlessSession {
3953
4155
  continue
3954
4156
  }
3955
4157
 
4158
+ // Evidence-gated completion (fabrication fix, 2026-09-01 — see
4159
+ // src/engine/claims.ts's module doc for the full writeup): when
4160
+ // evidenceRequiredCompletion is on, every machine-checkable claim
4161
+ // in the completion's result text (a file exists, a specific
4162
+ // command passed, serial markers appear, a PR exists) must be
4163
+ // independently verified against ground truth — the filesystem,
4164
+ // a real re-run of the exact command, the newest serial log, and
4165
+ // real git history — BEFORE the completion is accepted. This is
4166
+ // the structural backstop for the FINAL_REPORT's central finding:
4167
+ // a session claimed "all three hard gates pass" with a fabricated
4168
+ // serial-log excerpt when the driver was never merged and the
4169
+ // claimed Makefile target didn't exist. Fail-closed: any
4170
+ // unverifiable claim defers the completion with a corrective
4171
+ // message naming the specific claim, and emits an
4172
+ // `unverified_claim` feed event so downstream consumers (eval,
4173
+ // selfplay miner, orchestrator) can see WHY the completion was
4174
+ // not accepted. A result with NO machine-checkable claims (pure
4175
+ // prose) does not gate — but it also can never *pass* a gate.
4176
+ this.lastCompletionVerification = undefined
4177
+ if (this.config.evidenceRequiredCompletion) {
4178
+ const claims = extractClaims(result)
4179
+ if (claims.length > 0) {
4180
+ const verification = await verifyClaims(claims, {
4181
+ workspaceRoot: this.config.workspaceRoot,
4182
+ permissions: this.executor.permissions,
4183
+ })
4184
+ this.lastCompletionVerification = {
4185
+ claimsChecked: claims.length,
4186
+ claimsPassed: verification.filter((v) => v.verified).length,
4187
+ claimsUnverified: verification.filter((v) => !v.verified).length,
4188
+ }
4189
+ if (!allClaimsVerified(verification)) {
4190
+ // Identical-call guardrail reset — the same live
4191
+ // failure the verifyBeforeCompletion deferral above
4192
+ // documents (joeos issue #26, rounds 17 + 20): the
4193
+ // `continue` at the end of this block skips the
4194
+ // identical-call streak update below (it's part of
4195
+ // the normal per-iteration `calls` processing this
4196
+ // branch exits before reaching), so without this
4197
+ // reset lastCallBatchSignature/lastCallBatchNames/
4198
+ // identicalCallStreak stay FROZEN at whatever they
4199
+ // were when this deferral first started firing —
4200
+ // typically two identical execute_command failures
4201
+ // in a row, which is often exactly what triggers a
4202
+ // fabricated-completion deferral in the first
4203
+ // place (a model re-calling the same failing gate).
4204
+ // Every subsequent deferred-completion turn then
4205
+ // re-enters the request-prep cooldown-refresh loop
4206
+ // with identicalCallGuardActive still true and
4207
+ // lastCallBatchNames still ["execute_command"],
4208
+ // re-arming that tool's exclusion to the full
4209
+ // cooldown value EVERY turn before it ever ticks
4210
+ // down — the model has no legal move
4211
+ // (attempt_completion deferred, execute_command
4212
+ // excluded) and just keeps re-calling
4213
+ // attempt_completion, which is exactly the input
4214
+ // that keeps re-triggering this same `continue`
4215
+ // path. Confirmed live: 30+ iterations spinning
4216
+ // between "attempt_completion deferred" and an
4217
+ // unchanging "excludedToolCooldowns:
4218
+ // {execute_command: 4}" until the iteration cap was
4219
+ // hit. A deferred completion is definitionally not
4220
+ // a repeat of whatever tool-call batch came before
4221
+ // it, so the guard has no reason to stay active
4222
+ // into the next turn.
4223
+ lastCallBatchSignature = null
4224
+ lastCallBatchNames = []
4225
+ identicalCallStreak = 0
4226
+ identicalCallNudgeInjected = false
4227
+ excludedToolCooldowns.delete("execute_command")
4228
+ const firstBad = verification.find((v) => !v.verified)
4229
+ const unverifiedLabels = verification
4230
+ .filter((v) => !v.verified)
4231
+ .map((v) => claimLabel(v.claim))
4232
+ .join(", ")
4233
+ for (const sibling of calls) {
4234
+ if (sibling.id === completionCall.id) {
4235
+ continue
4236
+ }
4237
+ messages.push({
4238
+ role: "tool",
4239
+ tool_call_id: sibling.id,
4240
+ name: sibling.name,
4241
+ content:
4242
+ "[System: not executed — attempt_completion was deferred because your result makes claims that could not be verified against the real workspace; re-issue this call if still needed.]",
4243
+ })
4244
+ }
4245
+ messages.push({
4246
+ role: "tool",
4247
+ tool_call_id: completionCall.id,
4248
+ name: "attempt_completion",
4249
+ content:
4250
+ "[System: attempt_completion was NOT accepted. Your result claims: " +
4251
+ `${unverifiedLabels}. None of these could be independently confirmed: ` +
4252
+ `${firstBad?.detail ?? "no evidence found"}. ` +
4253
+ "Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
4254
+ "Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and re-issue attempt_completion, " +
4255
+ "or restate the result to only claim what you have actually verified.]",
4256
+ })
4257
+ this.logger.warn("[loop] attempt_completion deferred — unverifiable claims in result", {
4258
+ iteration,
4259
+ claims,
4260
+ verification: verification.map((v) => ({ verified: v.verified, detail: v.detail })),
4261
+ })
4262
+ this.scheduleAux(() =>
4263
+ this.emitEvent(
4264
+ "unverified_claim",
4265
+ () =>
4266
+ this.eventFeed.unverifiedClaim({
4267
+ iteration,
4268
+ claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
4269
+ claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
4270
+ claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
4271
+ detail: firstBad?.detail ?? "",
4272
+ }),
4273
+ { iteration },
4274
+ ),
4275
+ )
4276
+ continue
4277
+ }
4278
+ }
4279
+ }
4280
+
3956
4281
  this.logger.info("[loop] attempt_completion received — success", { iteration })
3957
4282
  const reportPath = await this.persistFinalReport(iteration, result)
3958
- return { status: "success", result, iterations: iteration, toolCalls: toolCalls + 1, reportPath }
4283
+ return {
4284
+ status: "success",
4285
+ result,
4286
+ iterations: iteration,
4287
+ toolCalls: toolCalls + 1,
4288
+ reportPath,
4289
+ verification: this.lastCompletionVerification,
4290
+ }
3959
4291
  }
3960
4292
 
3961
4293
  // 6b. Text-only reply (no tool_calls) → pragmatic success fallback,
@@ -4082,9 +4414,93 @@ export class HeadlessSession {
4082
4414
  artifactRejectionStreak = 0
4083
4415
  artifactRejectionNudgeInjected = false
4084
4416
  if (text && !this.config.requireExplicitCompletion) {
4417
+ // Evidence-gated completion (fabrication fix, 2026-09-01)
4418
+ // — the text-only success fallback is a REAL bypass for
4419
+ // cloud sessions: requireExplicitCompletion defaults OFF
4420
+ // for the cloud backend, so with --require-evidence a
4421
+ // cloud model could dump prose ("all three hard gates
4422
+ // pass…") and be recorded as success with ZERO
4423
+ // evidence, exactly the fabrication shape the gate
4424
+ // exists to stop. When evidenceRequiredCompletion is
4425
+ // ON, a text-only reply is treated as a completion
4426
+ // CANDIDATE and runs the SAME extract/verify gate as an
4427
+ // attempt_completion: pure prose (no machine-checkable
4428
+ // claims) or fully-verified claims are accepted; any
4429
+ // unverifiable claim defers with the corrective nudge
4430
+ // and an unverified_claim event, never a success.
4431
+ this.lastCompletionVerification = undefined
4432
+ if (this.config.evidenceRequiredCompletion) {
4433
+ const claims = extractClaims(text)
4434
+ if (claims.length > 0) {
4435
+ const verification = await verifyClaims(claims, {
4436
+ workspaceRoot: this.config.workspaceRoot,
4437
+ permissions: this.executor.permissions,
4438
+ })
4439
+ this.lastCompletionVerification = {
4440
+ claimsChecked: claims.length,
4441
+ claimsPassed: verification.filter((v) => v.verified).length,
4442
+ claimsUnverified: verification.filter((v) => !v.verified).length,
4443
+ }
4444
+ if (!allClaimsVerified(verification)) {
4445
+ // Same identical-call guardrail reset as the
4446
+ // attempt_completion deferral above — a text-only
4447
+ // reply is definitionally not a repeat of the
4448
+ // previous tool-call batch.
4449
+ lastCallBatchSignature = null
4450
+ lastCallBatchNames = []
4451
+ identicalCallStreak = 0
4452
+ identicalCallNudgeInjected = false
4453
+ excludedToolCooldowns.delete("execute_command")
4454
+ const firstBad = verification.find((v) => !v.verified)
4455
+ const unverifiedLabels = verification
4456
+ .filter((v) => !v.verified)
4457
+ .map((v) => claimLabel(v.claim))
4458
+ .join(", ")
4459
+ messages.push({
4460
+ role: "user",
4461
+ content:
4462
+ `[System: your text-only reply was NOT accepted as a completion. It claims: ${unverifiedLabels}. ` +
4463
+ `None of these could be independently confirmed: ${firstBad?.detail ?? "no evidence found"}. ` +
4464
+ "Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
4465
+ "Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and then call attempt_completion, " +
4466
+ "or restate your answer to only claim what you have actually verified.]",
4467
+ })
4468
+ this.logger.warn("[loop] text-only reply NOT accepted — unverifiable claims", {
4469
+ iteration,
4470
+ claims,
4471
+ verification: verification.map((v) => ({
4472
+ verified: v.verified,
4473
+ detail: v.detail,
4474
+ })),
4475
+ })
4476
+ this.scheduleAux(() =>
4477
+ this.emitEvent(
4478
+ "unverified_claim",
4479
+ () =>
4480
+ this.eventFeed.unverifiedClaim({
4481
+ iteration,
4482
+ claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
4483
+ claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
4484
+ claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
4485
+ detail: firstBad?.detail ?? "",
4486
+ }),
4487
+ { iteration },
4488
+ ),
4489
+ )
4490
+ continue
4491
+ }
4492
+ }
4493
+ }
4085
4494
  this.logger.info("[loop] text-only reply (no tool calls) — success", { iteration })
4086
4495
  const reportPath = await this.persistFinalReport(iteration, text)
4087
- return { status: "success", result: text, iterations: iteration, toolCalls, reportPath }
4496
+ return {
4497
+ status: "success",
4498
+ result: text,
4499
+ iterations: iteration,
4500
+ toolCalls,
4501
+ reportPath,
4502
+ ...(this.lastCompletionVerification ? { verification: this.lastCompletionVerification } : {}),
4503
+ }
4088
4504
  }
4089
4505
  // Empty reply, or a text reply that requireExplicitCompletion
4090
4506
  // refuses to treat as final: nudge and count as a mistake.
@@ -4296,8 +4712,35 @@ export class HeadlessSession {
4296
4712
  ? resultContent.slice(0, 500)
4297
4713
  : JSON.stringify(resultContent).slice(0, 500)
4298
4714
  lastWriteToolSummary = `${call.name} ${target}\n${output}`
4715
+ lastWriteToolFailedTarget = toolCallPathArg(this.config.workspaceRoot, call)
4299
4716
  } else {
4300
4717
  lastWriteToolSummary = undefined
4718
+ lastWriteToolFailedTarget = undefined
4719
+ // A genuinely successful write resolves the situation for
4720
+ // real — the re-read-only clearing no longer needs to gate
4721
+ // anything (see writeFailedClearedByReReadOnly's doc above).
4722
+ writeFailedClearedByReReadOnly = false
4723
+ reReadClearedWriteSummary = undefined
4724
+ }
4725
+ }
4726
+ // See lastWriteToolFailedTarget's doc comment above: a
4727
+ // successful read_file of the EXACT file a write tool just
4728
+ // failed to edit is real re-verification evidence, not just
4729
+ // time passing — clear the stale-failure flag so a
4730
+ // subsequent honest completion (including "no edit was
4731
+ // actually needed") isn't blocked by a failure the model
4732
+ // has since genuinely re-checked.
4733
+ if (call.name === "read_file" && !isError && lastWriteToolFailed) {
4734
+ const readTarget = toolCallPathArg(this.config.workspaceRoot, call)
4735
+ if (readTarget !== undefined && readTarget === lastWriteToolFailedTarget) {
4736
+ lastWriteToolFailed = false
4737
+ reReadClearedWriteSummary = lastWriteToolSummary
4738
+ lastWriteToolSummary = undefined
4739
+ lastWriteToolFailedTarget = undefined
4740
+ // Remember this was cleared by a re-read ONLY, not by a
4741
+ // real successful write — completion still has to prove
4742
+ // it isn't claiming an edit it never landed.
4743
+ writeFailedClearedByReReadOnly = true
4301
4744
  }
4302
4745
  }
4303
4746
  const targetPath = editToolTargetPath(this.config.workspaceRoot, call)
@@ -4802,6 +5245,7 @@ export class HeadlessSession {
4802
5245
  patchLocalToolSchemas: this.config.patchLocalToolSchemas,
4803
5246
  verifyBeforeCompletion: this.config.verifyBeforeCompletion,
4804
5247
  guardLargeOverwrites: this.config.guardLargeOverwrites,
5248
+ disableReadFileCache: this.config.disableReadFileCache,
4805
5249
  llmClient: this.llmClient,
4806
5250
  logger: this.logger,
4807
5251
  memory: this.config.memory,
@@ -5311,9 +5755,20 @@ export function summarizeToolArg(name: string, args: Record<string, unknown>): s
5311
5755
  case "list_files":
5312
5756
  case "write_to_file":
5313
5757
  case "apply_diff":
5758
+ return str(args.path)
5314
5759
  case "search_replace":
5315
5760
  case "edit_file":
5316
- return str(args.path)
5761
+ // 2026-09-02: real, confirmed bug -- these two tools' native
5762
+ // schemas (src/vendor/zoo-code/.../native-tools/edit_file.ts,
5763
+ // search_replace.ts) declare `file_path`, not `path` (unlike
5764
+ // write_to_file/apply_diff/list_files, which really do use
5765
+ // `path`) -- so this always returned undefined -> "" here,
5766
+ // making the tool_call event's path summary silently blank for
5767
+ // every edit_file/search_replace call. That's exactly what made
5768
+ // live log-watching during a real session unable to show which
5769
+ // file was being edited. Fall back to `path` too in case an
5770
+ // older/alias caller still sends that key.
5771
+ return str(args.file_path) ?? str(args.path)
5317
5772
  case "execute_command":
5318
5773
  return str(args.command)?.slice(0, 200)
5319
5774
  case "update_todo_list":
@@ -660,7 +660,11 @@ export function buildHeadlessConventions(mode: string, customModes: ModeConfig[]
660
660
  - Commit before finishing: real, working changes must be committed via \`git add\` + \`git commit\` BEFORE calling attempt_completion. Use one commit per logical change with a descriptive message, matching this repo's normal style (run \`git log --oneline\` for examples). If you genuinely have no changes to commit (e.g. a read-only investigation), finish without committing.
661
661
  - Test selection during iterative work: after editing a file, run the SPECIFIC test file(s) for what you changed (the \`run_tests\` tool, or \`npx tsx <test-file>\` directly) instead of the full suite on every edit. The full \`npm test\` is still required once, right before attempt_completion, for real confidence.
662
662
  - Batch same-file diffs: when several changes target the SAME file, include them as separate SEARCH/REPLACE blocks in ONE apply_diff call — and after any successful edit, re-read the file before composing further diffs, because its content has changed.
663
- - Waiting on an external check (CI run, registry, live service) is legitimate verification work, but every wait call still counts against your iteration budget: prefer ONE long-running command with an explicit \`timeout\` — e.g. \`gh run watch <id> --interval 30 --exit-status\` — over repeated \`sleep N && gh run list\` polls. (Issue #119: polling burns iterations with no token spend.)`
663
+ - Waiting on an external check (CI run, registry, live service) is legitimate verification work, but every wait call still counts against your iteration budget: prefer ONE long-running command with an explicit \`timeout\` — e.g. \`gh run watch <id> --interval 30 --exit-status\` — over repeated \`sleep N && gh run list\` polls. (Issue #119: polling burns iterations with no token spend.)
664
+ - Check the real workspace before assuming a path exists: when a task resembles something you've worked on before, use \`list_files\`/\`read_file\` on THIS workspace first rather than guessing a path from a similar prior task — a plausible-looking guess against the wrong repo/scratch layout still fails, and repeating the same wrong guess wastes the iteration budget without new information.
665
+ - If the same command fails twice in a row, change what you do next — read the real error, try something different, or (if there's genuinely no path forward) report the honest state via \`attempt_completion\` rather than retrying the identical call or repeating the same diagnosis in prose without acting on it. A real, honest "this doesn't work and here's why" is always a valid, complete answer; a stalled loop is not.
666
+ - When \`attempt_completion\` is deferred (rejected because a check hasn't actually passed yet, an edit hasn't succeeded, or a claim isn't backed by real evidence), your VERY NEXT action must be the real fix the deferral message describes — a new command, a corrected edit, or gathering the missing evidence — never another \`attempt_completion\` call on its own. Retrying completion without first taking that corrective action just repeats the same rejection and burns the iteration budget with no new information; it will keep being deferred for the identical reason until you actually address it.
667
+ - Starting a long-running process (a dev server, a background listener) and then checking it — do this as TWO separate \`execute_command\` calls, not one. First call: run the command DIRECTLY, with no trailing \`&\` and no \`bash -c '... &'\` wrapper — e.g. \`python3 server.py\` with \`timeout: 2\`. The tool's own contract already keeps it running in the background past the timeout (you get the output captured so far, not an error); it does NOT need you to background it yourself. Second, separate call: the actual check (\`curl ...\`, \`nc -z ...\`). Manually backgrounding with \`&\` inside a \`bash -c\` (or any subshell) is a real, observed failure mode — the process gets backgrounded INSIDE that subshell, which then exits once the \`&\` returns, and the process can be reaped along with it before your next command in the same chain ever reaches it. If a check right after starting something fails, don't keep editing the file you just started — first suspect the START command's shape.`
664
668
  }
665
669
 
666
670
  // ─── Tool selection ──────────────────────────────────────────────────────────