headlesscode 1.0.3 → 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -62,6 +62,14 @@ import type { MemoryStore, RecallResult } from "../memory/types.js"
62
62
  import { createCheckpointService, type CheckpointService } from "../checkpoints/service.js"
63
63
  import { recordSessionUsage, removeLiveUsage, writeLiveUsage } from "./usage.js"
64
64
  import { EventFeed, EVENT_TRUNCATE_CHARS, truncateField } from "./events.js"
65
+ import {
66
+ allClaimsVerified,
67
+ claimLabel,
68
+ extractClaims,
69
+ firstUnverifiedDetail,
70
+ verifyClaims,
71
+ type ClaimVerification,
72
+ } from "./claims.js"
65
73
  import { writeSessionReport } from "./reports.js"
66
74
  import { Logger } from "./logger.js"
67
75
  import { extractEmbeddedToolCall, parseToolCalls } from "./parser.js"
@@ -89,6 +97,7 @@ import type {
89
97
  ParsedToolCall,
90
98
  SessionResult,
91
99
  SessionBudgetUsage,
100
+ SessionCompletionVerification,
92
101
  ToolContext,
93
102
  ToolResult,
94
103
  } from "./types.js"
@@ -1068,6 +1077,37 @@ export interface HeadlessSessionConfig {
1068
1077
  * code-mode backend turns this on.
1069
1078
  */
1070
1079
  verifyBeforeCompletion?: boolean
1080
+ /**
1081
+ * Evidence-gated completion — the fabrication fix (2026-09-01, see
1082
+ * src/engine/claims.ts). When true, `attempt_completion` is refused
1083
+ * unless every machine-checkable claim its result text makes (a file
1084
+ * exists, a specific command passed, serial markers appear, a PR
1085
+ * exists) is independently verified against ground truth:
1086
+ *
1087
+ * - file claim → fs.stat on the resolved workspace path
1088
+ * - command claim → RE-RUN the exact command (same permission gate as
1089
+ * execute_command), require exit 0
1090
+ * - serial marker → grep the newest build/serial-*.log for the claimed
1091
+ * ordered markers
1092
+ * - PR claim → require real git-history evidence of the number
1093
+ *
1094
+ * Ground truth comes from the filesystem and real re-runs — never from
1095
+ * the model's own prose. This is the structural backstop for the exact
1096
+ * failure the FINAL_REPORT documented (§4): a session claimed "all
1097
+ * three hard gates pass" with a fabricated serial-log excerpt when the
1098
+ * driver was never merged and the claimed Makefile target didn't exist.
1099
+ * Fail-closed: any unverifiable claim defers the completion with a
1100
+ * corrective message naming the specific unverified claim.
1101
+ *
1102
+ * Deliberately separate from verifyBeforeCompletion (which is
1103
+ * token/state-based and only looks at the LAST command): this checks
1104
+ * the CONTENT of the completion's claims against the real world,
1105
+ * independent of what the session did or didn't run before.
1106
+ *
1107
+ * Default false; opt-in via --require-evidence or the local code
1108
+ * backend (see cli.ts).
1109
+ */
1110
+ evidenceRequiredCompletion?: boolean
1071
1111
  /**
1072
1112
  * When false, cost is never tracked/accumulated for this session (see
1073
1113
  * BudgetTrackerOptions.trackCost) — local-backend sessions have no real
@@ -1159,6 +1199,24 @@ export interface HeadlessSessionConfig {
1159
1199
  * the local Ollama code-mode backend turns this on.
1160
1200
  */
1161
1201
  guardLargeOverwrites?: boolean
1202
+ /**
1203
+ * read_file/list_files carry a session-scoped "[cache] unchanged, reuse
1204
+ * the earlier result" short-circuit (src/tools/executor.ts) that saves
1205
+ * real, measured token cost against a remote model's per-token bill.
1206
+ * Verified live 2026-09-02 against the local backend: an edit_file
1207
+ * failure told a session to re-read and retry; it DID call read_file
1208
+ * again exactly as instructed, got the cache-hit notice instead of real
1209
+ * content (correct per the mechanism's own design — the safety valve
1210
+ * is "a SECOND consecutive identical call serves real content again"),
1211
+ * never made that second call, and fabricated an attempt_completion
1212
+ * instead. Re-serving a few hundred lines of file content costs a
1213
+ * local session near-nothing (prefill, not generation, against a GPU
1214
+ * with no per-token price) — cheap insurance locally against a much
1215
+ * more expensive failure mode. Default false; the local Ollama code-mode
1216
+ * backend turns this on (same opt-in shape as guardLargeOverwrites
1217
+ * above).
1218
+ */
1219
+ disableReadFileCache?: boolean
1162
1220
  /**
1163
1221
  * Phase 3 context condensation: the model's real context window in
1164
1222
  * tokens, used to decide WHEN to condense (the last request's real
@@ -1346,6 +1404,8 @@ export interface ResolvedSessionConfig {
1346
1404
  patchLocalToolSchemas: boolean
1347
1405
  /** See HeadlessSessionConfig.verifyBeforeCompletion (default false). */
1348
1406
  verifyBeforeCompletion: boolean
1407
+ /** See HeadlessSessionConfig.evidenceRequiredCompletion (default false). */
1408
+ evidenceRequiredCompletion: boolean
1349
1409
  /** See HeadlessSessionConfig.trackCost (default true). */
1350
1410
  trackCost: boolean
1351
1411
  /** See HeadlessSessionConfig.requireArtifactBeforeCompletion (default false). */
@@ -1358,6 +1418,8 @@ export interface ResolvedSessionConfig {
1358
1418
  requireArtifactSections?: string[]
1359
1419
  /** See HeadlessSessionConfig.guardLargeOverwrites (default false). */
1360
1420
  guardLargeOverwrites: boolean
1421
+ /** See HeadlessSessionConfig.disableReadFileCache (default false). */
1422
+ disableReadFileCache: boolean
1361
1423
  /**
1362
1424
  * Phase 3 context condensation: the model's real context window in
1363
1425
  * tokens, when explicitly configured (undefined = resolve live from
@@ -1536,6 +1598,17 @@ export class HeadlessSession {
1536
1598
  * BudgetTracker either way.
1537
1599
  */
1538
1600
  private sessionEnded = false
1601
+ /**
1602
+ * Evidence-gated completion (fabrication fix, 2026-09-01): the outcome of
1603
+ * the last attempt_completion claim-verification pass, when
1604
+ * evidenceRequiredCompletion was on and the result contained
1605
+ * machine-checkable claims. Carried onto the SessionResult (see
1606
+ * src/engine/types.ts SessionResult.verification) so callers can
1607
+ * distinguish "success claim independently verified" from "success
1608
+ * accepted on prose alone". undefined when the gate didn't run (flag off,
1609
+ * or no claims to check).
1610
+ */
1611
+ private lastCompletionVerification: SessionCompletionVerification | undefined = undefined
1539
1612
  /**
1540
1613
  * True once the real context window has been resolved (from config, the
1541
1614
  * OpenRouter models endpoint, or the conservative default) — the lookup
@@ -1680,12 +1753,14 @@ export class HeadlessSession {
1680
1753
  requireExplicitCompletion: config.requireExplicitCompletion ?? false,
1681
1754
  patchLocalToolSchemas: config.patchLocalToolSchemas ?? false,
1682
1755
  verifyBeforeCompletion: config.verifyBeforeCompletion ?? false,
1756
+ evidenceRequiredCompletion: config.evidenceRequiredCompletion ?? false,
1683
1757
  trackCost: config.trackCost ?? true,
1684
1758
  requireArtifactBeforeCompletion: config.requireArtifactBeforeCompletion ?? false,
1685
1759
  requireArtifactPathPattern: config.requireArtifactPathPattern,
1686
1760
  requireArtifactMinCitations: config.requireArtifactMinCitations,
1687
1761
  requireArtifactSections: config.requireArtifactSections,
1688
1762
  guardLargeOverwrites: config.guardLargeOverwrites ?? false,
1763
+ disableReadFileCache: config.disableReadFileCache ?? false,
1689
1764
  // Deliberately left undefined when the caller didn't configure it:
1690
1765
  // the live OpenRouter models-endpoint lookup in maybeCondenseHistory
1691
1766
  // resolves the real context window (never hardcode a stale number —
@@ -1735,6 +1810,7 @@ export class HeadlessSession {
1735
1810
  decisionPollIntervalMs: this.config.decisionPollIntervalMs,
1736
1811
  permissions: this.config.permissions,
1737
1812
  guardLargeOverwrites: this.config.guardLargeOverwrites,
1813
+ disableReadFileCache: this.config.disableReadFileCache,
1738
1814
  // Live worker monitoring: mirror ask_followup_question's
1739
1815
  // .harness.needs-decision marker lifecycle on the session's
1740
1816
  // event feed (decision_blocked / decision_answered). Non-fatal.
@@ -3242,6 +3318,57 @@ export class HeadlessSession {
3242
3318
  // write_to_file — see the doc comment where it's set for why this
3243
3319
  // exists as a separate flag.
3244
3320
  let lastWriteToolFailed = false
3321
+ // 2026-09-02: real, confirmed bug -- lastWriteToolFailed only ever
3322
+ // gets updated when edit_file/write_to_file/set_indentation is
3323
+ // called AGAIN, so once ANY such call fails, this stayed
3324
+ // permanently true for the REST of the session if the model never
3325
+ // touched that tool again -- even when it correctly determined, by
3326
+ // re-reading the real file, that no further edit was needed at
3327
+ // all. Verified live: a session read the file, correctly
3328
+ // concluded the /health handler it was asked to add already
3329
+ // existed and worked, and every subsequent honest, accurate
3330
+ // attempt_completion was deferred anyway with a stale "fix the
3331
+ // issue, make the edit succeed" message that no longer applied --
3332
+ // 15+ consecutive identical deferrals with no bounded-failure
3333
+ // kill-switch to end it (would have spun to the iteration cap).
3334
+ // Tracks the failed call's resolved target path so a genuine
3335
+ // re-verification (a successful read_file of that SAME file)
3336
+ // clears the flag -- narrow and hard to game (it requires actually
3337
+ // re-reading the exact file that failed to edit), unlike a blanket
3338
+ // reset on any successful tool call of any kind.
3339
+ let lastWriteToolFailedTarget: string | undefined
3340
+ // 2026-09-02 (same-day follow-up to the re-read clearing above): the
3341
+ // re-read escape hatch clears lastWriteToolFailed on the assumption
3342
+ // that a model which re-reads the file it failed to edit has
3343
+ // concluded "no edit was needed" and is about to finish honestly.
3344
+ // Verified live (followthrough sweep, "add an entry to a JSON array"
3345
+ // case): a session's edit_file failed, it re-read tasks.json exactly
3346
+ // as the re-read hatch expects, then called attempt_completion
3347
+ // claiming *"I added {\"id\": 7, \"name\": \"lint\"} to the tasks
3348
+ // array in tasks.json."* — an edit it never actually landed. The
3349
+ // re-read had cleared the flag, so the completion sailed through and
3350
+ // the sweep scored it a fabrication. The re-read alone can't tell
3351
+ // "no edit needed" (legit) from "edit still not made" (fabrication);
3352
+ // the completion prose is the discriminator. This flag records that
3353
+ // a failed write was cleared ONLY by a re-read (never by a real
3354
+ // successful write since); at completion time, if the result also
3355
+ // asserts an edit was made, the completion is deferred. A later
3356
+ // genuinely successful write tool call clears it for good.
3357
+ let writeFailedClearedByReReadOnly = false
3358
+ let reReadClearedWriteSummary: string | undefined
3359
+ // Second layer of defense alongside the lastWriteToolFailedTarget
3360
+ // fix above: even a LEGITIMATE reason to keep deferring (a real,
3361
+ // still-unfixed failure) has no bounded-failure kill-switch on this
3362
+ // specific path today — verified live it can spin to the full
3363
+ // iteration cap on unchanging identical deferrals with zero new
3364
+ // information, the same failure shape as issue #26's fix already
3365
+ // handled for the identical-tool-call-streak case but never for
3366
+ // this one. Counts consecutive verifyBeforeCompletion deferrals
3367
+ // carrying the SAME reason string; resets whenever the reason
3368
+ // changes (a changing reason means real progress/new information is
3369
+ // happening) or a real tool call succeeds.
3370
+ let consecutiveDeferralReason: string | undefined
3371
+ let consecutiveDeferralStreak = 0
3245
3372
  // See HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
3246
3373
  // comment (issue #143). Set true the first time this session calls
3247
3374
  // execute_command, write_to_file, or edit_file — regardless of
@@ -3249,6 +3376,23 @@ export class HeadlessSession {
3249
3376
  // tried to produce a real effect; lastExecuteCommandFailed/
3250
3377
  // lastWriteToolFailed above separately catch a FAILED attempt).
3251
3378
  let hasCalledArtifactTool = false
3379
+ // 2026-09-02: verified live in the fabrication sweep's e1000 case —
3380
+ // the requireArtifactBeforeCompletion deferral (below) has no
3381
+ // bounded-failure kill-switch of its own, unlike the
3382
+ // verifyBeforeCompletion deferral (consecutiveDeferralStreak) and the
3383
+ // identical-call streak (issue #26). A session that never once calls
3384
+ // a real artifact tool, keeps re-issuing attempt_completion, and
3385
+ // generates a large inline "report" each turn spun 17+ iterations /
3386
+ // 971s before the duration cap finally killed it — the model even
3387
+ // correctly diagnosed its own fabrication ("I never actually called
3388
+ // write_to_file… the report I gave was the real output I expected to
3389
+ // see AFTER running the gate") and kept going anyway. Count
3390
+ // consecutive hits of that specific deferral; once it reaches
3391
+ // consecutiveErrorLimit, end as a bounded failure. No reset needed:
3392
+ // the branch only fires while hasCalledArtifactTool is false, and the
3393
+ // first real execute_command/write_to_file/edit_file call (even a
3394
+ // failed one) flips that permanently.
3395
+ let artifactBeforeCompletionDeferrals = 0
3252
3396
  // Concrete command + truncated error output for the failure above,
3253
3397
  // so the attempt_completion rejection below can restate WHAT failed
3254
3398
  // instead of pointing at it abstractly — by the time a model reaches
@@ -3632,10 +3776,17 @@ export class HeadlessSession {
3632
3776
  siblingCount: calls.length - 1,
3633
3777
  })
3634
3778
  } else if (completionCall) {
3635
- const result =
3636
- typeof completionCall.args?.result === "string"
3637
- ? completionCall.args.result
3638
- : JSON.stringify(completionCall.args)
3779
+ const rawResult = completionCall.args?.result
3780
+ if (typeof rawResult !== "string" || rawResult.trim() === "") {
3781
+ return this.boundedFailure(
3782
+ "malformed attempt_completion result",
3783
+ iteration,
3784
+ toolCalls + 1,
3785
+ 1,
3786
+ 1,
3787
+ )
3788
+ }
3789
+ const result = rawResult
3639
3790
 
3640
3791
  // Verify-before-finishing guardrail (see
3641
3792
  // HeadlessSessionConfig.verifyBeforeCompletion's doc comment): the
@@ -3677,9 +3828,30 @@ export class HeadlessSession {
3677
3828
  !messages.some(
3678
3829
  (m) => m.role === "tool" && m.name === "execute_command" && /\d/.test(String(m.content ?? "")),
3679
3830
  )
3831
+ // staleReReadEditClaim: a failed write was cleared by a re-read
3832
+ // ONLY (never a real successful write since — see
3833
+ // writeFailedClearedByReReadOnly's doc), yet this completion's prose
3834
+ // still asserts an edit was actually made. The re-read hatch exists
3835
+ // for the honest "no edit was needed" finish; a completion that
3836
+ // claims it *did* edit contradicts the still-unlanded write and is
3837
+ // deferred. Deliberately narrow: only affirmative "I/we added|
3838
+ // wrote|created|…", "added|inserted|… <x> (in)to <file>", or "the
3839
+ // edit/change has been made/applied" shapes — "already exists",
3840
+ // "no edit needed", "was already correct" carry no change verb and
3841
+ // still pass cleanly.
3842
+ const completionAssertsEditMade =
3843
+ /\b(?:I|we|I've|we've|I have|we have)\s+(?:just\s+|now\s+|successfully\s+)?(?:added|inserted|appended|wrote|written|created|updated|edited|modified|applied|replaced|changed)\b/i.test(
3844
+ result,
3845
+ ) ||
3846
+ /\b(?:added|inserted|appended|wrote|created|placed)\s+[^.\n]{0,80}?\b(?:in)?to\b\s+\S*[A-Za-z0-9_-]/i.test(result) ||
3847
+ /\bthe\s+(?:edit|change|fix|update|modification|entry|line|function|field)\s+(?:was|has been|is now)\s+(?:made|applied|written|added|inserted|in place|complete)\b/i.test(
3848
+ result,
3849
+ ) ||
3850
+ /\bhas been\s+(?:added|inserted|appended|written|updated|applied|modified|replaced)\b/i.test(result)
3851
+ const staleReReadEditClaim = writeFailedClearedByReReadOnly && completionAssertsEditMade
3680
3852
  if (
3681
3853
  this.config.verifyBeforeCompletion &&
3682
- (lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim)
3854
+ (lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim || staleReReadEditClaim)
3683
3855
  ) {
3684
3856
  // Verified live 2026-08-29 (joeos issue #26, rounds 17 + 20):
3685
3857
  // the `continue` at the end of this block skips the
@@ -3716,7 +3888,9 @@ export class HeadlessSession {
3716
3888
  identicalCallStreak = 0
3717
3889
  identicalCallNudgeInjected = false
3718
3890
  excludedToolCooldowns.delete("execute_command")
3719
- const reason = unsupportedMeasurementClaim
3891
+ const reason = staleReReadEditClaim
3892
+ ? "edit claimed but never landed (only a re-read since the failed write)"
3893
+ : unsupportedMeasurementClaim
3720
3894
  ? "unsupported measurement claim"
3721
3895
  : lastExecuteCommandFailed
3722
3896
  ? "last execute_command failed"
@@ -3729,7 +3903,9 @@ export class HeadlessSession {
3729
3903
  role: "tool",
3730
3904
  tool_call_id: sibling.id,
3731
3905
  name: sibling.name,
3732
- content: unsupportedMeasurementClaim
3906
+ content: staleReReadEditClaim
3907
+ ? "[System: not executed — attempt_completion was deferred because your result claims an edit was made, but the edit_file/write_to_file call for it failed and you have only re-read the file since (never landed a successful write); re-issue this call if still needed.]"
3908
+ : unsupportedMeasurementClaim
3733
3909
  ? "[System: not executed — attempt_completion was deferred because it makes a specific measurement/benchmark claim with no execute_command output in this session's history containing any supporting number; re-issue this call if still needed.]"
3734
3910
  : lastExecuteCommandFailed
3735
3911
  ? "[System: not executed — attempt_completion was deferred because the last command you ran ended in an error; re-issue this call if still needed.]"
@@ -3740,7 +3916,11 @@ export class HeadlessSession {
3740
3916
  role: "tool",
3741
3917
  tool_call_id: completionCall.id,
3742
3918
  name: "attempt_completion",
3743
- content: unsupportedMeasurementClaim
3919
+ content: staleReReadEditClaim
3920
+ ? "[System: attempt_completion was NOT accepted. Your result describes an edit as made (e.g. \"added …\", \"the entry has been added\"), but the edit_file/write_to_file call for it failed and the only thing you have done since is re-read the file — no successful write ever landed:\n" +
3921
+ `${reReadClearedWriteSummary ?? "(edit output no longer available)"}\n` +
3922
+ "Either actually make the edit succeed and re-issue attempt_completion, or, if no edit was truly needed, restate the result to say so plainly (e.g. \"no change was required\") without claiming an edit you did not land.]"
3923
+ : unsupportedMeasurementClaim
3744
3924
  ? "[System: attempt_completion was NOT accepted. Your result claims a specific measurement/benchmark, but no execute_command output anywhere in this session contains a supporting number. Either run the real command that produces this evidence and re-issue attempt_completion, or restate the result without the unsupported claim.]"
3745
3925
  : lastExecuteCommandFailed
3746
3926
  ? "[System: attempt_completion was NOT accepted. The most recent command you ran ended in an error, and you have not run a command since that succeeded:\n" +
@@ -3751,8 +3931,28 @@ export class HeadlessSession {
3751
3931
  "Fix the issue, make the edit succeed, and only call attempt_completion again once it actually applied.]",
3752
3932
  })
3753
3933
  this.logger.warn("[loop] attempt_completion deferred", { iteration, reason })
3934
+ if (reason === consecutiveDeferralReason) {
3935
+ consecutiveDeferralStreak++
3936
+ } else {
3937
+ consecutiveDeferralReason = reason
3938
+ consecutiveDeferralStreak = 1
3939
+ }
3940
+ if (consecutiveDeferralStreak >= this.config.consecutiveErrorLimit) {
3941
+ return this.boundedFailure(
3942
+ "consecutive completion deferrals (same reason, no new evidence)",
3943
+ iteration,
3944
+ toolCalls,
3945
+ consecutiveDeferralStreak,
3946
+ )
3947
+ }
3754
3948
  continue
3755
3949
  }
3950
+ // A turn that reaches here without hitting the deferral branch
3951
+ // above represents real progress (a normal tool call ran, or
3952
+ // this SPECIFIC completion was actually accepted) — the streak
3953
+ // above only means anything as CONSECUTIVE identical deferrals.
3954
+ consecutiveDeferralReason = undefined
3955
+ consecutiveDeferralStreak = 0
3756
3956
 
3757
3957
  // requireArtifactBeforeCompletion guardrail (issue #143) — see
3758
3958
  // HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
@@ -3782,6 +3982,15 @@ export class HeadlessSession {
3782
3982
  iteration,
3783
3983
  reason: "no artifact-producing tool call in this session",
3784
3984
  })
3985
+ artifactBeforeCompletionDeferrals++
3986
+ if (artifactBeforeCompletionDeferrals >= this.config.consecutiveErrorLimit) {
3987
+ return this.boundedFailure(
3988
+ "repeated attempt_completion with no artifact-producing tool call ever made",
3989
+ iteration,
3990
+ toolCalls,
3991
+ artifactBeforeCompletionDeferrals,
3992
+ )
3993
+ }
3785
3994
  continue
3786
3995
  }
3787
3996
 
@@ -3953,9 +4162,139 @@ export class HeadlessSession {
3953
4162
  continue
3954
4163
  }
3955
4164
 
4165
+ // Evidence-gated completion (fabrication fix, 2026-09-01 — see
4166
+ // src/engine/claims.ts's module doc for the full writeup): when
4167
+ // evidenceRequiredCompletion is on, every machine-checkable claim
4168
+ // in the completion's result text (a file exists, a specific
4169
+ // command passed, serial markers appear, a PR exists) must be
4170
+ // independently verified against ground truth — the filesystem,
4171
+ // a real re-run of the exact command, the newest serial log, and
4172
+ // real git history — BEFORE the completion is accepted. This is
4173
+ // the structural backstop for the FINAL_REPORT's central finding:
4174
+ // a session claimed "all three hard gates pass" with a fabricated
4175
+ // serial-log excerpt when the driver was never merged and the
4176
+ // claimed Makefile target didn't exist. Fail-closed: any
4177
+ // unverifiable claim defers the completion with a corrective
4178
+ // message naming the specific claim, and emits an
4179
+ // `unverified_claim` feed event so downstream consumers (eval,
4180
+ // selfplay miner, orchestrator) can see WHY the completion was
4181
+ // not accepted. A result with NO machine-checkable claims (pure
4182
+ // prose) does not gate — but it also can never *pass* a gate.
4183
+ this.lastCompletionVerification = undefined
4184
+ if (this.config.evidenceRequiredCompletion) {
4185
+ const claims = extractClaims(result)
4186
+ if (claims.length > 0) {
4187
+ const verification = await verifyClaims(claims, {
4188
+ workspaceRoot: this.config.workspaceRoot,
4189
+ permissions: this.executor.permissions,
4190
+ })
4191
+ this.lastCompletionVerification = {
4192
+ claimsChecked: claims.length,
4193
+ claimsPassed: verification.filter((v) => v.verified).length,
4194
+ claimsUnverified: verification.filter((v) => !v.verified).length,
4195
+ }
4196
+ if (!allClaimsVerified(verification)) {
4197
+ // Identical-call guardrail reset — the same live
4198
+ // failure the verifyBeforeCompletion deferral above
4199
+ // documents (joeos issue #26, rounds 17 + 20): the
4200
+ // `continue` at the end of this block skips the
4201
+ // identical-call streak update below (it's part of
4202
+ // the normal per-iteration `calls` processing this
4203
+ // branch exits before reaching), so without this
4204
+ // reset lastCallBatchSignature/lastCallBatchNames/
4205
+ // identicalCallStreak stay FROZEN at whatever they
4206
+ // were when this deferral first started firing —
4207
+ // typically two identical execute_command failures
4208
+ // in a row, which is often exactly what triggers a
4209
+ // fabricated-completion deferral in the first
4210
+ // place (a model re-calling the same failing gate).
4211
+ // Every subsequent deferred-completion turn then
4212
+ // re-enters the request-prep cooldown-refresh loop
4213
+ // with identicalCallGuardActive still true and
4214
+ // lastCallBatchNames still ["execute_command"],
4215
+ // re-arming that tool's exclusion to the full
4216
+ // cooldown value EVERY turn before it ever ticks
4217
+ // down — the model has no legal move
4218
+ // (attempt_completion deferred, execute_command
4219
+ // excluded) and just keeps re-calling
4220
+ // attempt_completion, which is exactly the input
4221
+ // that keeps re-triggering this same `continue`
4222
+ // path. Confirmed live: 30+ iterations spinning
4223
+ // between "attempt_completion deferred" and an
4224
+ // unchanging "excludedToolCooldowns:
4225
+ // {execute_command: 4}" until the iteration cap was
4226
+ // hit. A deferred completion is definitionally not
4227
+ // a repeat of whatever tool-call batch came before
4228
+ // it, so the guard has no reason to stay active
4229
+ // into the next turn.
4230
+ lastCallBatchSignature = null
4231
+ lastCallBatchNames = []
4232
+ identicalCallStreak = 0
4233
+ identicalCallNudgeInjected = false
4234
+ excludedToolCooldowns.delete("execute_command")
4235
+ const firstBad = verification.find((v) => !v.verified)
4236
+ const unverifiedLabels = verification
4237
+ .filter((v) => !v.verified)
4238
+ .map((v) => claimLabel(v.claim))
4239
+ .join(", ")
4240
+ for (const sibling of calls) {
4241
+ if (sibling.id === completionCall.id) {
4242
+ continue
4243
+ }
4244
+ messages.push({
4245
+ role: "tool",
4246
+ tool_call_id: sibling.id,
4247
+ name: sibling.name,
4248
+ content:
4249
+ "[System: not executed — attempt_completion was deferred because your result makes claims that could not be verified against the real workspace; re-issue this call if still needed.]",
4250
+ })
4251
+ }
4252
+ messages.push({
4253
+ role: "tool",
4254
+ tool_call_id: completionCall.id,
4255
+ name: "attempt_completion",
4256
+ content:
4257
+ "[System: attempt_completion was NOT accepted. Your result claims: " +
4258
+ `${unverifiedLabels}. None of these could be independently confirmed: ` +
4259
+ `${firstBad?.detail ?? "no evidence found"}. ` +
4260
+ "Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
4261
+ "Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and re-issue attempt_completion, " +
4262
+ "or restate the result to only claim what you have actually verified.]",
4263
+ })
4264
+ this.logger.warn("[loop] attempt_completion deferred — unverifiable claims in result", {
4265
+ iteration,
4266
+ claims,
4267
+ verification: verification.map((v) => ({ verified: v.verified, detail: v.detail })),
4268
+ })
4269
+ this.scheduleAux(() =>
4270
+ this.emitEvent(
4271
+ "unverified_claim",
4272
+ () =>
4273
+ this.eventFeed.unverifiedClaim({
4274
+ iteration,
4275
+ claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
4276
+ claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
4277
+ claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
4278
+ detail: firstBad?.detail ?? "",
4279
+ }),
4280
+ { iteration },
4281
+ ),
4282
+ )
4283
+ continue
4284
+ }
4285
+ }
4286
+ }
4287
+
3956
4288
  this.logger.info("[loop] attempt_completion received — success", { iteration })
3957
4289
  const reportPath = await this.persistFinalReport(iteration, result)
3958
- return { status: "success", result, iterations: iteration, toolCalls: toolCalls + 1, reportPath }
4290
+ return {
4291
+ status: "success",
4292
+ result,
4293
+ iterations: iteration,
4294
+ toolCalls: toolCalls + 1,
4295
+ reportPath,
4296
+ verification: this.lastCompletionVerification,
4297
+ }
3959
4298
  }
3960
4299
 
3961
4300
  // 6b. Text-only reply (no tool_calls) → pragmatic success fallback,
@@ -4082,9 +4421,93 @@ export class HeadlessSession {
4082
4421
  artifactRejectionStreak = 0
4083
4422
  artifactRejectionNudgeInjected = false
4084
4423
  if (text && !this.config.requireExplicitCompletion) {
4424
+ // Evidence-gated completion (fabrication fix, 2026-09-01)
4425
+ // — the text-only success fallback is a REAL bypass for
4426
+ // cloud sessions: requireExplicitCompletion defaults OFF
4427
+ // for the cloud backend, so with --require-evidence a
4428
+ // cloud model could dump prose ("all three hard gates
4429
+ // pass…") and be recorded as success with ZERO
4430
+ // evidence, exactly the fabrication shape the gate
4431
+ // exists to stop. When evidenceRequiredCompletion is
4432
+ // ON, a text-only reply is treated as a completion
4433
+ // CANDIDATE and runs the SAME extract/verify gate as an
4434
+ // attempt_completion: pure prose (no machine-checkable
4435
+ // claims) or fully-verified claims are accepted; any
4436
+ // unverifiable claim defers with the corrective nudge
4437
+ // and an unverified_claim event, never a success.
4438
+ this.lastCompletionVerification = undefined
4439
+ if (this.config.evidenceRequiredCompletion) {
4440
+ const claims = extractClaims(text)
4441
+ if (claims.length > 0) {
4442
+ const verification = await verifyClaims(claims, {
4443
+ workspaceRoot: this.config.workspaceRoot,
4444
+ permissions: this.executor.permissions,
4445
+ })
4446
+ this.lastCompletionVerification = {
4447
+ claimsChecked: claims.length,
4448
+ claimsPassed: verification.filter((v) => v.verified).length,
4449
+ claimsUnverified: verification.filter((v) => !v.verified).length,
4450
+ }
4451
+ if (!allClaimsVerified(verification)) {
4452
+ // Same identical-call guardrail reset as the
4453
+ // attempt_completion deferral above — a text-only
4454
+ // reply is definitionally not a repeat of the
4455
+ // previous tool-call batch.
4456
+ lastCallBatchSignature = null
4457
+ lastCallBatchNames = []
4458
+ identicalCallStreak = 0
4459
+ identicalCallNudgeInjected = false
4460
+ excludedToolCooldowns.delete("execute_command")
4461
+ const firstBad = verification.find((v) => !v.verified)
4462
+ const unverifiedLabels = verification
4463
+ .filter((v) => !v.verified)
4464
+ .map((v) => claimLabel(v.claim))
4465
+ .join(", ")
4466
+ messages.push({
4467
+ role: "user",
4468
+ content:
4469
+ `[System: your text-only reply was NOT accepted as a completion. It claims: ${unverifiedLabels}. ` +
4470
+ `None of these could be independently confirmed: ${firstBad?.detail ?? "no evidence found"}. ` +
4471
+ "Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
4472
+ "Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and then call attempt_completion, " +
4473
+ "or restate your answer to only claim what you have actually verified.]",
4474
+ })
4475
+ this.logger.warn("[loop] text-only reply NOT accepted — unverifiable claims", {
4476
+ iteration,
4477
+ claims,
4478
+ verification: verification.map((v) => ({
4479
+ verified: v.verified,
4480
+ detail: v.detail,
4481
+ })),
4482
+ })
4483
+ this.scheduleAux(() =>
4484
+ this.emitEvent(
4485
+ "unverified_claim",
4486
+ () =>
4487
+ this.eventFeed.unverifiedClaim({
4488
+ iteration,
4489
+ claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
4490
+ claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
4491
+ claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
4492
+ detail: firstBad?.detail ?? "",
4493
+ }),
4494
+ { iteration },
4495
+ ),
4496
+ )
4497
+ continue
4498
+ }
4499
+ }
4500
+ }
4085
4501
  this.logger.info("[loop] text-only reply (no tool calls) — success", { iteration })
4086
4502
  const reportPath = await this.persistFinalReport(iteration, text)
4087
- return { status: "success", result: text, iterations: iteration, toolCalls, reportPath }
4503
+ return {
4504
+ status: "success",
4505
+ result: text,
4506
+ iterations: iteration,
4507
+ toolCalls,
4508
+ reportPath,
4509
+ ...(this.lastCompletionVerification ? { verification: this.lastCompletionVerification } : {}),
4510
+ }
4088
4511
  }
4089
4512
  // Empty reply, or a text reply that requireExplicitCompletion
4090
4513
  // refuses to treat as final: nudge and count as a mistake.
@@ -4296,8 +4719,35 @@ export class HeadlessSession {
4296
4719
  ? resultContent.slice(0, 500)
4297
4720
  : JSON.stringify(resultContent).slice(0, 500)
4298
4721
  lastWriteToolSummary = `${call.name} ${target}\n${output}`
4722
+ lastWriteToolFailedTarget = toolCallPathArg(this.config.workspaceRoot, call)
4299
4723
  } else {
4300
4724
  lastWriteToolSummary = undefined
4725
+ lastWriteToolFailedTarget = undefined
4726
+ // A genuinely successful write resolves the situation for
4727
+ // real — the re-read-only clearing no longer needs to gate
4728
+ // anything (see writeFailedClearedByReReadOnly's doc above).
4729
+ writeFailedClearedByReReadOnly = false
4730
+ reReadClearedWriteSummary = undefined
4731
+ }
4732
+ }
4733
+ // See lastWriteToolFailedTarget's doc comment above: a
4734
+ // successful read_file of the EXACT file a write tool just
4735
+ // failed to edit is real re-verification evidence, not just
4736
+ // time passing — clear the stale-failure flag so a
4737
+ // subsequent honest completion (including "no edit was
4738
+ // actually needed") isn't blocked by a failure the model
4739
+ // has since genuinely re-checked.
4740
+ if (call.name === "read_file" && !isError && lastWriteToolFailed) {
4741
+ const readTarget = toolCallPathArg(this.config.workspaceRoot, call)
4742
+ if (readTarget !== undefined && readTarget === lastWriteToolFailedTarget) {
4743
+ lastWriteToolFailed = false
4744
+ reReadClearedWriteSummary = lastWriteToolSummary
4745
+ lastWriteToolSummary = undefined
4746
+ lastWriteToolFailedTarget = undefined
4747
+ // Remember this was cleared by a re-read ONLY, not by a
4748
+ // real successful write — completion still has to prove
4749
+ // it isn't claiming an edit it never landed.
4750
+ writeFailedClearedByReReadOnly = true
4301
4751
  }
4302
4752
  }
4303
4753
  const targetPath = editToolTargetPath(this.config.workspaceRoot, call)
@@ -4802,6 +5252,7 @@ export class HeadlessSession {
4802
5252
  patchLocalToolSchemas: this.config.patchLocalToolSchemas,
4803
5253
  verifyBeforeCompletion: this.config.verifyBeforeCompletion,
4804
5254
  guardLargeOverwrites: this.config.guardLargeOverwrites,
5255
+ disableReadFileCache: this.config.disableReadFileCache,
4805
5256
  llmClient: this.llmClient,
4806
5257
  logger: this.logger,
4807
5258
  memory: this.config.memory,
@@ -5311,9 +5762,20 @@ export function summarizeToolArg(name: string, args: Record<string, unknown>): s
5311
5762
  case "list_files":
5312
5763
  case "write_to_file":
5313
5764
  case "apply_diff":
5765
+ return str(args.path)
5314
5766
  case "search_replace":
5315
5767
  case "edit_file":
5316
- return str(args.path)
5768
+ // 2026-09-02: real, confirmed bug -- these two tools' native
5769
+ // schemas (src/vendor/zoo-code/.../native-tools/edit_file.ts,
5770
+ // search_replace.ts) declare `file_path`, not `path` (unlike
5771
+ // write_to_file/apply_diff/list_files, which really do use
5772
+ // `path`) -- so this always returned undefined -> "" here,
5773
+ // making the tool_call event's path summary silently blank for
5774
+ // every edit_file/search_replace call. That's exactly what made
5775
+ // live log-watching during a real session unable to show which
5776
+ // file was being edited. Fall back to `path` too in case an
5777
+ // older/alias caller still sends that key.
5778
+ return str(args.file_path) ?? str(args.path)
5317
5779
  case "execute_command":
5318
5780
  return str(args.command)?.slice(0, 200)
5319
5781
  case "update_todo_list":