headlesscode 1.0.3 → 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -56
- package/package.json +1 -1
- package/src/cli.ts +115 -4
- package/src/engine/claims.ts +503 -0
- package/src/engine/events.ts +32 -0
- package/src/engine/lazy-tools.ts +30 -9
- package/src/engine/loop.ts +473 -11
- package/src/engine/prompt.ts +5 -1
- package/src/engine/types.ts +36 -0
- package/src/rsi/archive.ts +129 -0
- package/src/rsi/config.ts +312 -0
- package/src/rsi/controller.ts +268 -0
- package/src/rsi/curriculum.ts +68 -0
- package/src/rsi/evaluator.ts +106 -0
- package/src/rsi/fitness.ts +64 -0
- package/src/rsi/index.ts +16 -0
- package/src/rsi/models.ts +89 -0
- package/src/rsi/mutation.ts +77 -0
- package/src/rsi/reports.ts +47 -0
- package/src/rsi/roles.ts +37 -0
- package/src/rsi/sandbox.ts +10 -0
- package/src/rsi/search.ts +32 -0
- package/src/rsi/selection.ts +132 -0
- package/src/rsi/trajectory.ts +143 -0
- package/src/rsi/types.ts +317 -0
- package/src/rsi/workspace.ts +96 -0
- package/src/tools/executor.ts +79 -3
package/src/engine/loop.ts
CHANGED
|
@@ -62,6 +62,14 @@ import type { MemoryStore, RecallResult } from "../memory/types.js"
|
|
|
62
62
|
import { createCheckpointService, type CheckpointService } from "../checkpoints/service.js"
|
|
63
63
|
import { recordSessionUsage, removeLiveUsage, writeLiveUsage } from "./usage.js"
|
|
64
64
|
import { EventFeed, EVENT_TRUNCATE_CHARS, truncateField } from "./events.js"
|
|
65
|
+
import {
|
|
66
|
+
allClaimsVerified,
|
|
67
|
+
claimLabel,
|
|
68
|
+
extractClaims,
|
|
69
|
+
firstUnverifiedDetail,
|
|
70
|
+
verifyClaims,
|
|
71
|
+
type ClaimVerification,
|
|
72
|
+
} from "./claims.js"
|
|
65
73
|
import { writeSessionReport } from "./reports.js"
|
|
66
74
|
import { Logger } from "./logger.js"
|
|
67
75
|
import { extractEmbeddedToolCall, parseToolCalls } from "./parser.js"
|
|
@@ -89,6 +97,7 @@ import type {
|
|
|
89
97
|
ParsedToolCall,
|
|
90
98
|
SessionResult,
|
|
91
99
|
SessionBudgetUsage,
|
|
100
|
+
SessionCompletionVerification,
|
|
92
101
|
ToolContext,
|
|
93
102
|
ToolResult,
|
|
94
103
|
} from "./types.js"
|
|
@@ -1068,6 +1077,37 @@ export interface HeadlessSessionConfig {
|
|
|
1068
1077
|
* code-mode backend turns this on.
|
|
1069
1078
|
*/
|
|
1070
1079
|
verifyBeforeCompletion?: boolean
|
|
1080
|
+
/**
|
|
1081
|
+
* Evidence-gated completion — the fabrication fix (2026-09-01, see
|
|
1082
|
+
* src/engine/claims.ts). When true, `attempt_completion` is refused
|
|
1083
|
+
* unless every machine-checkable claim its result text makes (a file
|
|
1084
|
+
* exists, a specific command passed, serial markers appear, a PR
|
|
1085
|
+
* exists) is independently verified against ground truth:
|
|
1086
|
+
*
|
|
1087
|
+
* - file claim → fs.stat on the resolved workspace path
|
|
1088
|
+
* - command claim → RE-RUN the exact command (same permission gate as
|
|
1089
|
+
* execute_command), require exit 0
|
|
1090
|
+
* - serial marker → grep the newest build/serial-*.log for the claimed
|
|
1091
|
+
* ordered markers
|
|
1092
|
+
* - PR claim → require real git-history evidence of the number
|
|
1093
|
+
*
|
|
1094
|
+
* Ground truth comes from the filesystem and real re-runs — never from
|
|
1095
|
+
* the model's own prose. This is the structural backstop for the exact
|
|
1096
|
+
* failure the FINAL_REPORT documented (§4): a session claimed "all
|
|
1097
|
+
* three hard gates pass" with a fabricated serial-log excerpt when the
|
|
1098
|
+
* driver was never merged and the claimed Makefile target didn't exist.
|
|
1099
|
+
* Fail-closed: any unverifiable claim defers the completion with a
|
|
1100
|
+
* corrective message naming the specific unverified claim.
|
|
1101
|
+
*
|
|
1102
|
+
* Deliberately separate from verifyBeforeCompletion (which is
|
|
1103
|
+
* token/state-based and only looks at the LAST command): this checks
|
|
1104
|
+
* the CONTENT of the completion's claims against the real world,
|
|
1105
|
+
* independent of what the session did or didn't run before.
|
|
1106
|
+
*
|
|
1107
|
+
* Default false; opt-in via --require-evidence or the local code
|
|
1108
|
+
* backend (see cli.ts).
|
|
1109
|
+
*/
|
|
1110
|
+
evidenceRequiredCompletion?: boolean
|
|
1071
1111
|
/**
|
|
1072
1112
|
* When false, cost is never tracked/accumulated for this session (see
|
|
1073
1113
|
* BudgetTrackerOptions.trackCost) — local-backend sessions have no real
|
|
@@ -1159,6 +1199,24 @@ export interface HeadlessSessionConfig {
|
|
|
1159
1199
|
* the local Ollama code-mode backend turns this on.
|
|
1160
1200
|
*/
|
|
1161
1201
|
guardLargeOverwrites?: boolean
|
|
1202
|
+
/**
|
|
1203
|
+
* read_file/list_files carry a session-scoped "[cache] unchanged, reuse
|
|
1204
|
+
* the earlier result" short-circuit (src/tools/executor.ts) that saves
|
|
1205
|
+
* real, measured token cost against a remote model's per-token bill.
|
|
1206
|
+
* Verified live 2026-09-02 against the local backend: an edit_file
|
|
1207
|
+
* failure told a session to re-read and retry; it DID call read_file
|
|
1208
|
+
* again exactly as instructed, got the cache-hit notice instead of real
|
|
1209
|
+
* content (correct per the mechanism's own design — the safety valve
|
|
1210
|
+
* is "a SECOND consecutive identical call serves real content again"),
|
|
1211
|
+
* never made that second call, and fabricated an attempt_completion
|
|
1212
|
+
* instead. Re-serving a few hundred lines of file content costs a
|
|
1213
|
+
* local session near-nothing (prefill, not generation, against a GPU
|
|
1214
|
+
* with no per-token price) — cheap insurance locally against a much
|
|
1215
|
+
* more expensive failure mode. Default false; the local Ollama code-mode
|
|
1216
|
+
* backend turns this on (same opt-in shape as guardLargeOverwrites
|
|
1217
|
+
* above).
|
|
1218
|
+
*/
|
|
1219
|
+
disableReadFileCache?: boolean
|
|
1162
1220
|
/**
|
|
1163
1221
|
* Phase 3 context condensation: the model's real context window in
|
|
1164
1222
|
* tokens, used to decide WHEN to condense (the last request's real
|
|
@@ -1346,6 +1404,8 @@ export interface ResolvedSessionConfig {
|
|
|
1346
1404
|
patchLocalToolSchemas: boolean
|
|
1347
1405
|
/** See HeadlessSessionConfig.verifyBeforeCompletion (default false). */
|
|
1348
1406
|
verifyBeforeCompletion: boolean
|
|
1407
|
+
/** See HeadlessSessionConfig.evidenceRequiredCompletion (default false). */
|
|
1408
|
+
evidenceRequiredCompletion: boolean
|
|
1349
1409
|
/** See HeadlessSessionConfig.trackCost (default true). */
|
|
1350
1410
|
trackCost: boolean
|
|
1351
1411
|
/** See HeadlessSessionConfig.requireArtifactBeforeCompletion (default false). */
|
|
@@ -1358,6 +1418,8 @@ export interface ResolvedSessionConfig {
|
|
|
1358
1418
|
requireArtifactSections?: string[]
|
|
1359
1419
|
/** See HeadlessSessionConfig.guardLargeOverwrites (default false). */
|
|
1360
1420
|
guardLargeOverwrites: boolean
|
|
1421
|
+
/** See HeadlessSessionConfig.disableReadFileCache (default false). */
|
|
1422
|
+
disableReadFileCache: boolean
|
|
1361
1423
|
/**
|
|
1362
1424
|
* Phase 3 context condensation: the model's real context window in
|
|
1363
1425
|
* tokens, when explicitly configured (undefined = resolve live from
|
|
@@ -1536,6 +1598,17 @@ export class HeadlessSession {
|
|
|
1536
1598
|
* BudgetTracker either way.
|
|
1537
1599
|
*/
|
|
1538
1600
|
private sessionEnded = false
|
|
1601
|
+
/**
|
|
1602
|
+
* Evidence-gated completion (fabrication fix, 2026-09-01): the outcome of
|
|
1603
|
+
* the last attempt_completion claim-verification pass, when
|
|
1604
|
+
* evidenceRequiredCompletion was on and the result contained
|
|
1605
|
+
* machine-checkable claims. Carried onto the SessionResult (see
|
|
1606
|
+
* src/engine/types.ts SessionResult.verification) so callers can
|
|
1607
|
+
* distinguish "success claim independently verified" from "success
|
|
1608
|
+
* accepted on prose alone". undefined when the gate didn't run (flag off,
|
|
1609
|
+
* or no claims to check).
|
|
1610
|
+
*/
|
|
1611
|
+
private lastCompletionVerification: SessionCompletionVerification | undefined = undefined
|
|
1539
1612
|
/**
|
|
1540
1613
|
* True once the real context window has been resolved (from config, the
|
|
1541
1614
|
* OpenRouter models endpoint, or the conservative default) — the lookup
|
|
@@ -1680,12 +1753,14 @@ export class HeadlessSession {
|
|
|
1680
1753
|
requireExplicitCompletion: config.requireExplicitCompletion ?? false,
|
|
1681
1754
|
patchLocalToolSchemas: config.patchLocalToolSchemas ?? false,
|
|
1682
1755
|
verifyBeforeCompletion: config.verifyBeforeCompletion ?? false,
|
|
1756
|
+
evidenceRequiredCompletion: config.evidenceRequiredCompletion ?? false,
|
|
1683
1757
|
trackCost: config.trackCost ?? true,
|
|
1684
1758
|
requireArtifactBeforeCompletion: config.requireArtifactBeforeCompletion ?? false,
|
|
1685
1759
|
requireArtifactPathPattern: config.requireArtifactPathPattern,
|
|
1686
1760
|
requireArtifactMinCitations: config.requireArtifactMinCitations,
|
|
1687
1761
|
requireArtifactSections: config.requireArtifactSections,
|
|
1688
1762
|
guardLargeOverwrites: config.guardLargeOverwrites ?? false,
|
|
1763
|
+
disableReadFileCache: config.disableReadFileCache ?? false,
|
|
1689
1764
|
// Deliberately left undefined when the caller didn't configure it:
|
|
1690
1765
|
// the live OpenRouter models-endpoint lookup in maybeCondenseHistory
|
|
1691
1766
|
// resolves the real context window (never hardcode a stale number —
|
|
@@ -1735,6 +1810,7 @@ export class HeadlessSession {
|
|
|
1735
1810
|
decisionPollIntervalMs: this.config.decisionPollIntervalMs,
|
|
1736
1811
|
permissions: this.config.permissions,
|
|
1737
1812
|
guardLargeOverwrites: this.config.guardLargeOverwrites,
|
|
1813
|
+
disableReadFileCache: this.config.disableReadFileCache,
|
|
1738
1814
|
// Live worker monitoring: mirror ask_followup_question's
|
|
1739
1815
|
// .harness.needs-decision marker lifecycle on the session's
|
|
1740
1816
|
// event feed (decision_blocked / decision_answered). Non-fatal.
|
|
@@ -3242,6 +3318,57 @@ export class HeadlessSession {
|
|
|
3242
3318
|
// write_to_file — see the doc comment where it's set for why this
|
|
3243
3319
|
// exists as a separate flag.
|
|
3244
3320
|
let lastWriteToolFailed = false
|
|
3321
|
+
// 2026-09-02: real, confirmed bug -- lastWriteToolFailed only ever
|
|
3322
|
+
// gets updated when edit_file/write_to_file/set_indentation is
|
|
3323
|
+
// called AGAIN, so once ANY such call fails, this stayed
|
|
3324
|
+
// permanently true for the REST of the session if the model never
|
|
3325
|
+
// touched that tool again -- even when it correctly determined, by
|
|
3326
|
+
// re-reading the real file, that no further edit was needed at
|
|
3327
|
+
// all. Verified live: a session read the file, correctly
|
|
3328
|
+
// concluded the /health handler it was asked to add already
|
|
3329
|
+
// existed and worked, and every subsequent honest, accurate
|
|
3330
|
+
// attempt_completion was deferred anyway with a stale "fix the
|
|
3331
|
+
// issue, make the edit succeed" message that no longer applied --
|
|
3332
|
+
// 15+ consecutive identical deferrals with no bounded-failure
|
|
3333
|
+
// kill-switch to end it (would have spun to the iteration cap).
|
|
3334
|
+
// Tracks the failed call's resolved target path so a genuine
|
|
3335
|
+
// re-verification (a successful read_file of that SAME file)
|
|
3336
|
+
// clears the flag -- narrow and hard to game (it requires actually
|
|
3337
|
+
// re-reading the exact file that failed to edit), unlike a blanket
|
|
3338
|
+
// reset on any successful tool call of any kind.
|
|
3339
|
+
let lastWriteToolFailedTarget: string | undefined
|
|
3340
|
+
// 2026-09-02 (same-day follow-up to the re-read clearing above): the
|
|
3341
|
+
// re-read escape hatch clears lastWriteToolFailed on the assumption
|
|
3342
|
+
// that a model which re-reads the file it failed to edit has
|
|
3343
|
+
// concluded "no edit was needed" and is about to finish honestly.
|
|
3344
|
+
// Verified live (followthrough sweep, "add an entry to a JSON array"
|
|
3345
|
+
// case): a session's edit_file failed, it re-read tasks.json exactly
|
|
3346
|
+
// as the re-read hatch expects, then called attempt_completion
|
|
3347
|
+
// claiming *"I added {\"id\": 7, \"name\": \"lint\"} to the tasks
|
|
3348
|
+
// array in tasks.json."* — an edit it never actually landed. The
|
|
3349
|
+
// re-read had cleared the flag, so the completion sailed through and
|
|
3350
|
+
// the sweep scored it a fabrication. The re-read alone can't tell
|
|
3351
|
+
// "no edit needed" (legit) from "edit still not made" (fabrication);
|
|
3352
|
+
// the completion prose is the discriminator. This flag records that
|
|
3353
|
+
// a failed write was cleared ONLY by a re-read (never by a real
|
|
3354
|
+
// successful write since); at completion time, if the result also
|
|
3355
|
+
// asserts an edit was made, the completion is deferred. A later
|
|
3356
|
+
// genuinely successful write tool call clears it for good.
|
|
3357
|
+
let writeFailedClearedByReReadOnly = false
|
|
3358
|
+
let reReadClearedWriteSummary: string | undefined
|
|
3359
|
+
// Second layer of defense alongside the lastWriteToolFailedTarget
|
|
3360
|
+
// fix above: even a LEGITIMATE reason to keep deferring (a real,
|
|
3361
|
+
// still-unfixed failure) has no bounded-failure kill-switch on this
|
|
3362
|
+
// specific path today — verified live it can spin to the full
|
|
3363
|
+
// iteration cap on unchanging identical deferrals with zero new
|
|
3364
|
+
// information, the same failure shape as issue #26's fix already
|
|
3365
|
+
// handled for the identical-tool-call-streak case but never for
|
|
3366
|
+
// this one. Counts consecutive verifyBeforeCompletion deferrals
|
|
3367
|
+
// carrying the SAME reason string; resets whenever the reason
|
|
3368
|
+
// changes (a changing reason means real progress/new information is
|
|
3369
|
+
// happening) or a real tool call succeeds.
|
|
3370
|
+
let consecutiveDeferralReason: string | undefined
|
|
3371
|
+
let consecutiveDeferralStreak = 0
|
|
3245
3372
|
// See HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
|
|
3246
3373
|
// comment (issue #143). Set true the first time this session calls
|
|
3247
3374
|
// execute_command, write_to_file, or edit_file — regardless of
|
|
@@ -3249,6 +3376,23 @@ export class HeadlessSession {
|
|
|
3249
3376
|
// tried to produce a real effect; lastExecuteCommandFailed/
|
|
3250
3377
|
// lastWriteToolFailed above separately catch a FAILED attempt).
|
|
3251
3378
|
let hasCalledArtifactTool = false
|
|
3379
|
+
// 2026-09-02: verified live in the fabrication sweep's e1000 case —
|
|
3380
|
+
// the requireArtifactBeforeCompletion deferral (below) has no
|
|
3381
|
+
// bounded-failure kill-switch of its own, unlike the
|
|
3382
|
+
// verifyBeforeCompletion deferral (consecutiveDeferralStreak) and the
|
|
3383
|
+
// identical-call streak (issue #26). A session that never once calls
|
|
3384
|
+
// a real artifact tool, keeps re-issuing attempt_completion, and
|
|
3385
|
+
// generates a large inline "report" each turn spun 17+ iterations /
|
|
3386
|
+
// 971s before the duration cap finally killed it — the model even
|
|
3387
|
+
// correctly diagnosed its own fabrication ("I never actually called
|
|
3388
|
+
// write_to_file… the report I gave was the real output I expected to
|
|
3389
|
+
// see AFTER running the gate") and kept going anyway. Count
|
|
3390
|
+
// consecutive hits of that specific deferral; once it reaches
|
|
3391
|
+
// consecutiveErrorLimit, end as a bounded failure. No reset needed:
|
|
3392
|
+
// the branch only fires while hasCalledArtifactTool is false, and the
|
|
3393
|
+
// first real execute_command/write_to_file/edit_file call (even a
|
|
3394
|
+
// failed one) flips that permanently.
|
|
3395
|
+
let artifactBeforeCompletionDeferrals = 0
|
|
3252
3396
|
// Concrete command + truncated error output for the failure above,
|
|
3253
3397
|
// so the attempt_completion rejection below can restate WHAT failed
|
|
3254
3398
|
// instead of pointing at it abstractly — by the time a model reaches
|
|
@@ -3632,10 +3776,17 @@ export class HeadlessSession {
|
|
|
3632
3776
|
siblingCount: calls.length - 1,
|
|
3633
3777
|
})
|
|
3634
3778
|
} else if (completionCall) {
|
|
3635
|
-
const
|
|
3636
|
-
|
|
3637
|
-
|
|
3638
|
-
|
|
3779
|
+
const rawResult = completionCall.args?.result
|
|
3780
|
+
if (typeof rawResult !== "string" || rawResult.trim() === "") {
|
|
3781
|
+
return this.boundedFailure(
|
|
3782
|
+
"malformed attempt_completion result",
|
|
3783
|
+
iteration,
|
|
3784
|
+
toolCalls + 1,
|
|
3785
|
+
1,
|
|
3786
|
+
1,
|
|
3787
|
+
)
|
|
3788
|
+
}
|
|
3789
|
+
const result = rawResult
|
|
3639
3790
|
|
|
3640
3791
|
// Verify-before-finishing guardrail (see
|
|
3641
3792
|
// HeadlessSessionConfig.verifyBeforeCompletion's doc comment): the
|
|
@@ -3677,9 +3828,30 @@ export class HeadlessSession {
|
|
|
3677
3828
|
!messages.some(
|
|
3678
3829
|
(m) => m.role === "tool" && m.name === "execute_command" && /\d/.test(String(m.content ?? "")),
|
|
3679
3830
|
)
|
|
3831
|
+
// staleReReadEditClaim: a failed write was cleared by a re-read
|
|
3832
|
+
// ONLY (never a real successful write since — see
|
|
3833
|
+
// writeFailedClearedByReReadOnly's doc), yet this completion's prose
|
|
3834
|
+
// still asserts an edit was actually made. The re-read hatch exists
|
|
3835
|
+
// for the honest "no edit was needed" finish; a completion that
|
|
3836
|
+
// claims it *did* edit contradicts the still-unlanded write and is
|
|
3837
|
+
// deferred. Deliberately narrow: only affirmative "I/we added|
|
|
3838
|
+
// wrote|created|…", "added|inserted|… <x> (in)to <file>", or "the
|
|
3839
|
+
// edit/change has been made/applied" shapes — "already exists",
|
|
3840
|
+
// "no edit needed", "was already correct" carry no change verb and
|
|
3841
|
+
// still pass cleanly.
|
|
3842
|
+
const completionAssertsEditMade =
|
|
3843
|
+
/\b(?:I|we|I've|we've|I have|we have)\s+(?:just\s+|now\s+|successfully\s+)?(?:added|inserted|appended|wrote|written|created|updated|edited|modified|applied|replaced|changed)\b/i.test(
|
|
3844
|
+
result,
|
|
3845
|
+
) ||
|
|
3846
|
+
/\b(?:added|inserted|appended|wrote|created|placed)\s+[^.\n]{0,80}?\b(?:in)?to\b\s+\S*[A-Za-z0-9_-]/i.test(result) ||
|
|
3847
|
+
/\bthe\s+(?:edit|change|fix|update|modification|entry|line|function|field)\s+(?:was|has been|is now)\s+(?:made|applied|written|added|inserted|in place|complete)\b/i.test(
|
|
3848
|
+
result,
|
|
3849
|
+
) ||
|
|
3850
|
+
/\bhas been\s+(?:added|inserted|appended|written|updated|applied|modified|replaced)\b/i.test(result)
|
|
3851
|
+
const staleReReadEditClaim = writeFailedClearedByReReadOnly && completionAssertsEditMade
|
|
3680
3852
|
if (
|
|
3681
3853
|
this.config.verifyBeforeCompletion &&
|
|
3682
|
-
(lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim)
|
|
3854
|
+
(lastExecuteCommandFailed || lastWriteToolFailed || unsupportedMeasurementClaim || staleReReadEditClaim)
|
|
3683
3855
|
) {
|
|
3684
3856
|
// Verified live 2026-08-29 (joeos issue #26, rounds 17 + 20):
|
|
3685
3857
|
// the `continue` at the end of this block skips the
|
|
@@ -3716,7 +3888,9 @@ export class HeadlessSession {
|
|
|
3716
3888
|
identicalCallStreak = 0
|
|
3717
3889
|
identicalCallNudgeInjected = false
|
|
3718
3890
|
excludedToolCooldowns.delete("execute_command")
|
|
3719
|
-
const reason =
|
|
3891
|
+
const reason = staleReReadEditClaim
|
|
3892
|
+
? "edit claimed but never landed (only a re-read since the failed write)"
|
|
3893
|
+
: unsupportedMeasurementClaim
|
|
3720
3894
|
? "unsupported measurement claim"
|
|
3721
3895
|
: lastExecuteCommandFailed
|
|
3722
3896
|
? "last execute_command failed"
|
|
@@ -3729,7 +3903,9 @@ export class HeadlessSession {
|
|
|
3729
3903
|
role: "tool",
|
|
3730
3904
|
tool_call_id: sibling.id,
|
|
3731
3905
|
name: sibling.name,
|
|
3732
|
-
content:
|
|
3906
|
+
content: staleReReadEditClaim
|
|
3907
|
+
? "[System: not executed — attempt_completion was deferred because your result claims an edit was made, but the edit_file/write_to_file call for it failed and you have only re-read the file since (never landed a successful write); re-issue this call if still needed.]"
|
|
3908
|
+
: unsupportedMeasurementClaim
|
|
3733
3909
|
? "[System: not executed — attempt_completion was deferred because it makes a specific measurement/benchmark claim with no execute_command output in this session's history containing any supporting number; re-issue this call if still needed.]"
|
|
3734
3910
|
: lastExecuteCommandFailed
|
|
3735
3911
|
? "[System: not executed — attempt_completion was deferred because the last command you ran ended in an error; re-issue this call if still needed.]"
|
|
@@ -3740,7 +3916,11 @@ export class HeadlessSession {
|
|
|
3740
3916
|
role: "tool",
|
|
3741
3917
|
tool_call_id: completionCall.id,
|
|
3742
3918
|
name: "attempt_completion",
|
|
3743
|
-
content:
|
|
3919
|
+
content: staleReReadEditClaim
|
|
3920
|
+
? "[System: attempt_completion was NOT accepted. Your result describes an edit as made (e.g. \"added …\", \"the entry has been added\"), but the edit_file/write_to_file call for it failed and the only thing you have done since is re-read the file — no successful write ever landed:\n" +
|
|
3921
|
+
`${reReadClearedWriteSummary ?? "(edit output no longer available)"}\n` +
|
|
3922
|
+
"Either actually make the edit succeed and re-issue attempt_completion, or, if no edit was truly needed, restate the result to say so plainly (e.g. \"no change was required\") without claiming an edit you did not land.]"
|
|
3923
|
+
: unsupportedMeasurementClaim
|
|
3744
3924
|
? "[System: attempt_completion was NOT accepted. Your result claims a specific measurement/benchmark, but no execute_command output anywhere in this session contains a supporting number. Either run the real command that produces this evidence and re-issue attempt_completion, or restate the result without the unsupported claim.]"
|
|
3745
3925
|
: lastExecuteCommandFailed
|
|
3746
3926
|
? "[System: attempt_completion was NOT accepted. The most recent command you ran ended in an error, and you have not run a command since that succeeded:\n" +
|
|
@@ -3751,8 +3931,28 @@ export class HeadlessSession {
|
|
|
3751
3931
|
"Fix the issue, make the edit succeed, and only call attempt_completion again once it actually applied.]",
|
|
3752
3932
|
})
|
|
3753
3933
|
this.logger.warn("[loop] attempt_completion deferred", { iteration, reason })
|
|
3934
|
+
if (reason === consecutiveDeferralReason) {
|
|
3935
|
+
consecutiveDeferralStreak++
|
|
3936
|
+
} else {
|
|
3937
|
+
consecutiveDeferralReason = reason
|
|
3938
|
+
consecutiveDeferralStreak = 1
|
|
3939
|
+
}
|
|
3940
|
+
if (consecutiveDeferralStreak >= this.config.consecutiveErrorLimit) {
|
|
3941
|
+
return this.boundedFailure(
|
|
3942
|
+
"consecutive completion deferrals (same reason, no new evidence)",
|
|
3943
|
+
iteration,
|
|
3944
|
+
toolCalls,
|
|
3945
|
+
consecutiveDeferralStreak,
|
|
3946
|
+
)
|
|
3947
|
+
}
|
|
3754
3948
|
continue
|
|
3755
3949
|
}
|
|
3950
|
+
// A turn that reaches here without hitting the deferral branch
|
|
3951
|
+
// above represents real progress (a normal tool call ran, or
|
|
3952
|
+
// this SPECIFIC completion was actually accepted) — the streak
|
|
3953
|
+
// above only means anything as CONSECUTIVE identical deferrals.
|
|
3954
|
+
consecutiveDeferralReason = undefined
|
|
3955
|
+
consecutiveDeferralStreak = 0
|
|
3756
3956
|
|
|
3757
3957
|
// requireArtifactBeforeCompletion guardrail (issue #143) — see
|
|
3758
3958
|
// HeadlessSessionConfig.requireArtifactBeforeCompletion's doc
|
|
@@ -3782,6 +3982,15 @@ export class HeadlessSession {
|
|
|
3782
3982
|
iteration,
|
|
3783
3983
|
reason: "no artifact-producing tool call in this session",
|
|
3784
3984
|
})
|
|
3985
|
+
artifactBeforeCompletionDeferrals++
|
|
3986
|
+
if (artifactBeforeCompletionDeferrals >= this.config.consecutiveErrorLimit) {
|
|
3987
|
+
return this.boundedFailure(
|
|
3988
|
+
"repeated attempt_completion with no artifact-producing tool call ever made",
|
|
3989
|
+
iteration,
|
|
3990
|
+
toolCalls,
|
|
3991
|
+
artifactBeforeCompletionDeferrals,
|
|
3992
|
+
)
|
|
3993
|
+
}
|
|
3785
3994
|
continue
|
|
3786
3995
|
}
|
|
3787
3996
|
|
|
@@ -3953,9 +4162,139 @@ export class HeadlessSession {
|
|
|
3953
4162
|
continue
|
|
3954
4163
|
}
|
|
3955
4164
|
|
|
4165
|
+
// Evidence-gated completion (fabrication fix, 2026-09-01 — see
|
|
4166
|
+
// src/engine/claims.ts's module doc for the full writeup): when
|
|
4167
|
+
// evidenceRequiredCompletion is on, every machine-checkable claim
|
|
4168
|
+
// in the completion's result text (a file exists, a specific
|
|
4169
|
+
// command passed, serial markers appear, a PR exists) must be
|
|
4170
|
+
// independently verified against ground truth — the filesystem,
|
|
4171
|
+
// a real re-run of the exact command, the newest serial log, and
|
|
4172
|
+
// real git history — BEFORE the completion is accepted. This is
|
|
4173
|
+
// the structural backstop for the FINAL_REPORT's central finding:
|
|
4174
|
+
// a session claimed "all three hard gates pass" with a fabricated
|
|
4175
|
+
// serial-log excerpt when the driver was never merged and the
|
|
4176
|
+
// claimed Makefile target didn't exist. Fail-closed: any
|
|
4177
|
+
// unverifiable claim defers the completion with a corrective
|
|
4178
|
+
// message naming the specific claim, and emits an
|
|
4179
|
+
// `unverified_claim` feed event so downstream consumers (eval,
|
|
4180
|
+
// selfplay miner, orchestrator) can see WHY the completion was
|
|
4181
|
+
// not accepted. A result with NO machine-checkable claims (pure
|
|
4182
|
+
// prose) does not gate — but it also can never *pass* a gate.
|
|
4183
|
+
this.lastCompletionVerification = undefined
|
|
4184
|
+
if (this.config.evidenceRequiredCompletion) {
|
|
4185
|
+
const claims = extractClaims(result)
|
|
4186
|
+
if (claims.length > 0) {
|
|
4187
|
+
const verification = await verifyClaims(claims, {
|
|
4188
|
+
workspaceRoot: this.config.workspaceRoot,
|
|
4189
|
+
permissions: this.executor.permissions,
|
|
4190
|
+
})
|
|
4191
|
+
this.lastCompletionVerification = {
|
|
4192
|
+
claimsChecked: claims.length,
|
|
4193
|
+
claimsPassed: verification.filter((v) => v.verified).length,
|
|
4194
|
+
claimsUnverified: verification.filter((v) => !v.verified).length,
|
|
4195
|
+
}
|
|
4196
|
+
if (!allClaimsVerified(verification)) {
|
|
4197
|
+
// Identical-call guardrail reset — the same live
|
|
4198
|
+
// failure the verifyBeforeCompletion deferral above
|
|
4199
|
+
// documents (joeos issue #26, rounds 17 + 20): the
|
|
4200
|
+
// `continue` at the end of this block skips the
|
|
4201
|
+
// identical-call streak update below (it's part of
|
|
4202
|
+
// the normal per-iteration `calls` processing this
|
|
4203
|
+
// branch exits before reaching), so without this
|
|
4204
|
+
// reset lastCallBatchSignature/lastCallBatchNames/
|
|
4205
|
+
// identicalCallStreak stay FROZEN at whatever they
|
|
4206
|
+
// were when this deferral first started firing —
|
|
4207
|
+
// typically two identical execute_command failures
|
|
4208
|
+
// in a row, which is often exactly what triggers a
|
|
4209
|
+
// fabricated-completion deferral in the first
|
|
4210
|
+
// place (a model re-calling the same failing gate).
|
|
4211
|
+
// Every subsequent deferred-completion turn then
|
|
4212
|
+
// re-enters the request-prep cooldown-refresh loop
|
|
4213
|
+
// with identicalCallGuardActive still true and
|
|
4214
|
+
// lastCallBatchNames still ["execute_command"],
|
|
4215
|
+
// re-arming that tool's exclusion to the full
|
|
4216
|
+
// cooldown value EVERY turn before it ever ticks
|
|
4217
|
+
// down — the model has no legal move
|
|
4218
|
+
// (attempt_completion deferred, execute_command
|
|
4219
|
+
// excluded) and just keeps re-calling
|
|
4220
|
+
// attempt_completion, which is exactly the input
|
|
4221
|
+
// that keeps re-triggering this same `continue`
|
|
4222
|
+
// path. Confirmed live: 30+ iterations spinning
|
|
4223
|
+
// between "attempt_completion deferred" and an
|
|
4224
|
+
// unchanging "excludedToolCooldowns:
|
|
4225
|
+
// {execute_command: 4}" until the iteration cap was
|
|
4226
|
+
// hit. A deferred completion is definitionally not
|
|
4227
|
+
// a repeat of whatever tool-call batch came before
|
|
4228
|
+
// it, so the guard has no reason to stay active
|
|
4229
|
+
// into the next turn.
|
|
4230
|
+
lastCallBatchSignature = null
|
|
4231
|
+
lastCallBatchNames = []
|
|
4232
|
+
identicalCallStreak = 0
|
|
4233
|
+
identicalCallNudgeInjected = false
|
|
4234
|
+
excludedToolCooldowns.delete("execute_command")
|
|
4235
|
+
const firstBad = verification.find((v) => !v.verified)
|
|
4236
|
+
const unverifiedLabels = verification
|
|
4237
|
+
.filter((v) => !v.verified)
|
|
4238
|
+
.map((v) => claimLabel(v.claim))
|
|
4239
|
+
.join(", ")
|
|
4240
|
+
for (const sibling of calls) {
|
|
4241
|
+
if (sibling.id === completionCall.id) {
|
|
4242
|
+
continue
|
|
4243
|
+
}
|
|
4244
|
+
messages.push({
|
|
4245
|
+
role: "tool",
|
|
4246
|
+
tool_call_id: sibling.id,
|
|
4247
|
+
name: sibling.name,
|
|
4248
|
+
content:
|
|
4249
|
+
"[System: not executed — attempt_completion was deferred because your result makes claims that could not be verified against the real workspace; re-issue this call if still needed.]",
|
|
4250
|
+
})
|
|
4251
|
+
}
|
|
4252
|
+
messages.push({
|
|
4253
|
+
role: "tool",
|
|
4254
|
+
tool_call_id: completionCall.id,
|
|
4255
|
+
name: "attempt_completion",
|
|
4256
|
+
content:
|
|
4257
|
+
"[System: attempt_completion was NOT accepted. Your result claims: " +
|
|
4258
|
+
`${unverifiedLabels}. None of these could be independently confirmed: ` +
|
|
4259
|
+
`${firstBad?.detail ?? "no evidence found"}. ` +
|
|
4260
|
+
"Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
|
|
4261
|
+
"Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and re-issue attempt_completion, " +
|
|
4262
|
+
"or restate the result to only claim what you have actually verified.]",
|
|
4263
|
+
})
|
|
4264
|
+
this.logger.warn("[loop] attempt_completion deferred — unverifiable claims in result", {
|
|
4265
|
+
iteration,
|
|
4266
|
+
claims,
|
|
4267
|
+
verification: verification.map((v) => ({ verified: v.verified, detail: v.detail })),
|
|
4268
|
+
})
|
|
4269
|
+
this.scheduleAux(() =>
|
|
4270
|
+
this.emitEvent(
|
|
4271
|
+
"unverified_claim",
|
|
4272
|
+
() =>
|
|
4273
|
+
this.eventFeed.unverifiedClaim({
|
|
4274
|
+
iteration,
|
|
4275
|
+
claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
|
|
4276
|
+
claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
|
|
4277
|
+
claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
|
|
4278
|
+
detail: firstBad?.detail ?? "",
|
|
4279
|
+
}),
|
|
4280
|
+
{ iteration },
|
|
4281
|
+
),
|
|
4282
|
+
)
|
|
4283
|
+
continue
|
|
4284
|
+
}
|
|
4285
|
+
}
|
|
4286
|
+
}
|
|
4287
|
+
|
|
3956
4288
|
this.logger.info("[loop] attempt_completion received — success", { iteration })
|
|
3957
4289
|
const reportPath = await this.persistFinalReport(iteration, result)
|
|
3958
|
-
return {
|
|
4290
|
+
return {
|
|
4291
|
+
status: "success",
|
|
4292
|
+
result,
|
|
4293
|
+
iterations: iteration,
|
|
4294
|
+
toolCalls: toolCalls + 1,
|
|
4295
|
+
reportPath,
|
|
4296
|
+
verification: this.lastCompletionVerification,
|
|
4297
|
+
}
|
|
3959
4298
|
}
|
|
3960
4299
|
|
|
3961
4300
|
// 6b. Text-only reply (no tool_calls) → pragmatic success fallback,
|
|
@@ -4082,9 +4421,93 @@ export class HeadlessSession {
|
|
|
4082
4421
|
artifactRejectionStreak = 0
|
|
4083
4422
|
artifactRejectionNudgeInjected = false
|
|
4084
4423
|
if (text && !this.config.requireExplicitCompletion) {
|
|
4424
|
+
// Evidence-gated completion (fabrication fix, 2026-09-01)
|
|
4425
|
+
// — the text-only success fallback is a REAL bypass for
|
|
4426
|
+
// cloud sessions: requireExplicitCompletion defaults OFF
|
|
4427
|
+
// for the cloud backend, so with --require-evidence a
|
|
4428
|
+
// cloud model could dump prose ("all three hard gates
|
|
4429
|
+
// pass…") and be recorded as success with ZERO
|
|
4430
|
+
// evidence, exactly the fabrication shape the gate
|
|
4431
|
+
// exists to stop. When evidenceRequiredCompletion is
|
|
4432
|
+
// ON, a text-only reply is treated as a completion
|
|
4433
|
+
// CANDIDATE and runs the SAME extract/verify gate as an
|
|
4434
|
+
// attempt_completion: pure prose (no machine-checkable
|
|
4435
|
+
// claims) or fully-verified claims are accepted; any
|
|
4436
|
+
// unverifiable claim defers with the corrective nudge
|
|
4437
|
+
// and an unverified_claim event, never a success.
|
|
4438
|
+
this.lastCompletionVerification = undefined
|
|
4439
|
+
if (this.config.evidenceRequiredCompletion) {
|
|
4440
|
+
const claims = extractClaims(text)
|
|
4441
|
+
if (claims.length > 0) {
|
|
4442
|
+
const verification = await verifyClaims(claims, {
|
|
4443
|
+
workspaceRoot: this.config.workspaceRoot,
|
|
4444
|
+
permissions: this.executor.permissions,
|
|
4445
|
+
})
|
|
4446
|
+
this.lastCompletionVerification = {
|
|
4447
|
+
claimsChecked: claims.length,
|
|
4448
|
+
claimsPassed: verification.filter((v) => v.verified).length,
|
|
4449
|
+
claimsUnverified: verification.filter((v) => !v.verified).length,
|
|
4450
|
+
}
|
|
4451
|
+
if (!allClaimsVerified(verification)) {
|
|
4452
|
+
// Same identical-call guardrail reset as the
|
|
4453
|
+
// attempt_completion deferral above — a text-only
|
|
4454
|
+
// reply is definitionally not a repeat of the
|
|
4455
|
+
// previous tool-call batch.
|
|
4456
|
+
lastCallBatchSignature = null
|
|
4457
|
+
lastCallBatchNames = []
|
|
4458
|
+
identicalCallStreak = 0
|
|
4459
|
+
identicalCallNudgeInjected = false
|
|
4460
|
+
excludedToolCooldowns.delete("execute_command")
|
|
4461
|
+
const firstBad = verification.find((v) => !v.verified)
|
|
4462
|
+
const unverifiedLabels = verification
|
|
4463
|
+
.filter((v) => !v.verified)
|
|
4464
|
+
.map((v) => claimLabel(v.claim))
|
|
4465
|
+
.join(", ")
|
|
4466
|
+
messages.push({
|
|
4467
|
+
role: "user",
|
|
4468
|
+
content:
|
|
4469
|
+
`[System: your text-only reply was NOT accepted as a completion. It claims: ${unverifiedLabels}. ` +
|
|
4470
|
+
`None of these could be independently confirmed: ${firstBad?.detail ?? "no evidence found"}. ` +
|
|
4471
|
+
"Ground truth comes from the real filesystem and real command re-runs — never from a written report. " +
|
|
4472
|
+
"Either run/verify the real thing (re-run the exact command, confirm the file actually exists on disk, check the real serial log) and then call attempt_completion, " +
|
|
4473
|
+
"or restate your answer to only claim what you have actually verified.]",
|
|
4474
|
+
})
|
|
4475
|
+
this.logger.warn("[loop] text-only reply NOT accepted — unverifiable claims", {
|
|
4476
|
+
iteration,
|
|
4477
|
+
claims,
|
|
4478
|
+
verification: verification.map((v) => ({
|
|
4479
|
+
verified: v.verified,
|
|
4480
|
+
detail: v.detail,
|
|
4481
|
+
})),
|
|
4482
|
+
})
|
|
4483
|
+
this.scheduleAux(() =>
|
|
4484
|
+
this.emitEvent(
|
|
4485
|
+
"unverified_claim",
|
|
4486
|
+
() =>
|
|
4487
|
+
this.eventFeed.unverifiedClaim({
|
|
4488
|
+
iteration,
|
|
4489
|
+
claimsChecked: this.lastCompletionVerification?.claimsChecked ?? 0,
|
|
4490
|
+
claimsPassed: this.lastCompletionVerification?.claimsPassed ?? 0,
|
|
4491
|
+
claimsUnverified: this.lastCompletionVerification?.claimsUnverified ?? 0,
|
|
4492
|
+
detail: firstBad?.detail ?? "",
|
|
4493
|
+
}),
|
|
4494
|
+
{ iteration },
|
|
4495
|
+
),
|
|
4496
|
+
)
|
|
4497
|
+
continue
|
|
4498
|
+
}
|
|
4499
|
+
}
|
|
4500
|
+
}
|
|
4085
4501
|
this.logger.info("[loop] text-only reply (no tool calls) — success", { iteration })
|
|
4086
4502
|
const reportPath = await this.persistFinalReport(iteration, text)
|
|
4087
|
-
return {
|
|
4503
|
+
return {
|
|
4504
|
+
status: "success",
|
|
4505
|
+
result: text,
|
|
4506
|
+
iterations: iteration,
|
|
4507
|
+
toolCalls,
|
|
4508
|
+
reportPath,
|
|
4509
|
+
...(this.lastCompletionVerification ? { verification: this.lastCompletionVerification } : {}),
|
|
4510
|
+
}
|
|
4088
4511
|
}
|
|
4089
4512
|
// Empty reply, or a text reply that requireExplicitCompletion
|
|
4090
4513
|
// refuses to treat as final: nudge and count as a mistake.
|
|
@@ -4296,8 +4719,35 @@ export class HeadlessSession {
|
|
|
4296
4719
|
? resultContent.slice(0, 500)
|
|
4297
4720
|
: JSON.stringify(resultContent).slice(0, 500)
|
|
4298
4721
|
lastWriteToolSummary = `${call.name} ${target}\n${output}`
|
|
4722
|
+
lastWriteToolFailedTarget = toolCallPathArg(this.config.workspaceRoot, call)
|
|
4299
4723
|
} else {
|
|
4300
4724
|
lastWriteToolSummary = undefined
|
|
4725
|
+
lastWriteToolFailedTarget = undefined
|
|
4726
|
+
// A genuinely successful write resolves the situation for
|
|
4727
|
+
// real — the re-read-only clearing no longer needs to gate
|
|
4728
|
+
// anything (see writeFailedClearedByReReadOnly's doc above).
|
|
4729
|
+
writeFailedClearedByReReadOnly = false
|
|
4730
|
+
reReadClearedWriteSummary = undefined
|
|
4731
|
+
}
|
|
4732
|
+
}
|
|
4733
|
+
// See lastWriteToolFailedTarget's doc comment above: a
|
|
4734
|
+
// successful read_file of the EXACT file a write tool just
|
|
4735
|
+
// failed to edit is real re-verification evidence, not just
|
|
4736
|
+
// time passing — clear the stale-failure flag so a
|
|
4737
|
+
// subsequent honest completion (including "no edit was
|
|
4738
|
+
// actually needed") isn't blocked by a failure the model
|
|
4739
|
+
// has since genuinely re-checked.
|
|
4740
|
+
if (call.name === "read_file" && !isError && lastWriteToolFailed) {
|
|
4741
|
+
const readTarget = toolCallPathArg(this.config.workspaceRoot, call)
|
|
4742
|
+
if (readTarget !== undefined && readTarget === lastWriteToolFailedTarget) {
|
|
4743
|
+
lastWriteToolFailed = false
|
|
4744
|
+
reReadClearedWriteSummary = lastWriteToolSummary
|
|
4745
|
+
lastWriteToolSummary = undefined
|
|
4746
|
+
lastWriteToolFailedTarget = undefined
|
|
4747
|
+
// Remember this was cleared by a re-read ONLY, not by a
|
|
4748
|
+
// real successful write — completion still has to prove
|
|
4749
|
+
// it isn't claiming an edit it never landed.
|
|
4750
|
+
writeFailedClearedByReReadOnly = true
|
|
4301
4751
|
}
|
|
4302
4752
|
}
|
|
4303
4753
|
const targetPath = editToolTargetPath(this.config.workspaceRoot, call)
|
|
@@ -4802,6 +5252,7 @@ export class HeadlessSession {
|
|
|
4802
5252
|
patchLocalToolSchemas: this.config.patchLocalToolSchemas,
|
|
4803
5253
|
verifyBeforeCompletion: this.config.verifyBeforeCompletion,
|
|
4804
5254
|
guardLargeOverwrites: this.config.guardLargeOverwrites,
|
|
5255
|
+
disableReadFileCache: this.config.disableReadFileCache,
|
|
4805
5256
|
llmClient: this.llmClient,
|
|
4806
5257
|
logger: this.logger,
|
|
4807
5258
|
memory: this.config.memory,
|
|
@@ -5311,9 +5762,20 @@ export function summarizeToolArg(name: string, args: Record<string, unknown>): s
|
|
|
5311
5762
|
case "list_files":
|
|
5312
5763
|
case "write_to_file":
|
|
5313
5764
|
case "apply_diff":
|
|
5765
|
+
return str(args.path)
|
|
5314
5766
|
case "search_replace":
|
|
5315
5767
|
case "edit_file":
|
|
5316
|
-
|
|
5768
|
+
// 2026-09-02: real, confirmed bug -- these two tools' native
|
|
5769
|
+
// schemas (src/vendor/zoo-code/.../native-tools/edit_file.ts,
|
|
5770
|
+
// search_replace.ts) declare `file_path`, not `path` (unlike
|
|
5771
|
+
// write_to_file/apply_diff/list_files, which really do use
|
|
5772
|
+
// `path`) -- so this always returned undefined -> "" here,
|
|
5773
|
+
// making the tool_call event's path summary silently blank for
|
|
5774
|
+
// every edit_file/search_replace call. That's exactly what made
|
|
5775
|
+
// live log-watching during a real session unable to show which
|
|
5776
|
+
// file was being edited. Fall back to `path` too in case an
|
|
5777
|
+
// older/alias caller still sends that key.
|
|
5778
|
+
return str(args.file_path) ?? str(args.path)
|
|
5317
5779
|
case "execute_command":
|
|
5318
5780
|
return str(args.command)?.slice(0, 200)
|
|
5319
5781
|
case "update_todo_list":
|