headlesscode 1.0.3 → 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -56
- package/package.json +1 -1
- package/src/cli.ts +115 -4
- package/src/engine/claims.ts +503 -0
- package/src/engine/events.ts +32 -0
- package/src/engine/lazy-tools.ts +30 -9
- package/src/engine/loop.ts +473 -11
- package/src/engine/prompt.ts +5 -1
- package/src/engine/types.ts +36 -0
- package/src/rsi/archive.ts +129 -0
- package/src/rsi/config.ts +312 -0
- package/src/rsi/controller.ts +268 -0
- package/src/rsi/curriculum.ts +68 -0
- package/src/rsi/evaluator.ts +106 -0
- package/src/rsi/fitness.ts +64 -0
- package/src/rsi/index.ts +16 -0
- package/src/rsi/models.ts +89 -0
- package/src/rsi/mutation.ts +77 -0
- package/src/rsi/reports.ts +47 -0
- package/src/rsi/roles.ts +37 -0
- package/src/rsi/sandbox.ts +10 -0
- package/src/rsi/search.ts +32 -0
- package/src/rsi/selection.ts +132 -0
- package/src/rsi/trajectory.ts +143 -0
- package/src/rsi/types.ts +317 -0
- package/src/rsi/workspace.ts +96 -0
- package/src/tools/executor.ts +79 -3
|
@@ -0,0 +1,503 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evidence-gated completion — the fabrication-fix core (2026-09-01).
|
|
3
|
+
*
|
|
4
|
+
* The FINAL_REPORT's central finding (joeos_finetune_data/FINAL_REPORT.md,
|
|
5
|
+
* §4): a real headlesscode session reported complete success with a
|
|
6
|
+
* fabricated QEMU serial log excerpt ("E1000: 1 / E1000: 2") claiming three
|
|
7
|
+
* hard gates passed, when the driver file was never merged and the claimed
|
|
8
|
+
* `make qemu-e1000-smoke` target did not even exist. Every pre-existing
|
|
9
|
+
* completion guardrail was token-based — it checked *wording* (did the last
|
|
10
|
+
* command fail? did any write tool ever get called?), never *ground truth*
|
|
11
|
+
* (does the file the model names exist with the content implied? did the
|
|
12
|
+
* exact command the model names actually run and pass?).
|
|
13
|
+
*
|
|
14
|
+
* This module closes that gap with the same philosophy the orchestrator's
|
|
15
|
+
* verification-gate.ts already applies to review/QA verdicts: the model's
|
|
16
|
+
* own prose is never a source of truth about what really happened. Ground
|
|
17
|
+
* truth comes from the filesystem and real re-runs, never from the report.
|
|
18
|
+
*
|
|
19
|
+
* Two stages:
|
|
20
|
+
*
|
|
21
|
+
* 1. `extractClaims` — conservatively parse an attempt_completion result
|
|
22
|
+
* for machine-checkable claims. ONLY claims that are BOTH specific
|
|
23
|
+
* (name a concrete file, command, marker, or PR) AND verifiable get
|
|
24
|
+
* extracted. Vague prose ("the driver works") is not a checkable claim
|
|
25
|
+
* and never gates — but it also can never *pass* a gate.
|
|
26
|
+
* 2. `verifyClaims` — check each extracted claim against ground truth:
|
|
27
|
+
* file_exists → fs.stat on the resolved path (path-safety enforced)
|
|
28
|
+
* command_passed→ RE-RUN the exact command through the same
|
|
29
|
+
* permission gate as execute_command, require exit 0
|
|
30
|
+
* (+ optional expected output marker)
|
|
31
|
+
* serial_marker → grep the newest build/serial-*.log for the claimed
|
|
32
|
+
* ordered markers (mirrors the qemu-smoke gate idiom)
|
|
33
|
+
* pr_url → only verified if the PR number appears in real
|
|
34
|
+
* git history (a real gh/git side effect); else
|
|
35
|
+
* fail-closed
|
|
36
|
+
*
|
|
37
|
+
* Fail-closed posture throughout: a claim that cannot be verified (missing
|
|
38
|
+
* file, failed re-run, no serial log, no git evidence) is UNVERIFIED, never
|
|
39
|
+
* "probably fine". The loop refuses the completion and names the specific
|
|
40
|
+
* unverified claim so the model has something concrete to fix.
|
|
41
|
+
*/
|
|
42
|
+
|
|
43
|
+
import * as fs from "node:fs"
|
|
44
|
+
import * as fsp from "node:fs/promises"
|
|
45
|
+
import * as path from "node:path"
|
|
46
|
+
import { execFile } from "node:child_process"
|
|
47
|
+
|
|
48
|
+
import { checkCommand, checkRedirectEscape } from "../permissions/commands.js"
|
|
49
|
+
import type { PermissionsConfig } from "../permissions/config.js"
|
|
50
|
+
import { BASH_PATH } from "../tools/executor.js"
|
|
51
|
+
|
|
52
|
+
/** A machine-checkable claim extracted from an attempt_completion result. */
|
|
53
|
+
export type ExtractedClaim =
|
|
54
|
+
| {
|
|
55
|
+
kind: "file_exists"
|
|
56
|
+
/** The workspace-relative path the result claims exists (posix). */
|
|
57
|
+
path: string
|
|
58
|
+
}
|
|
59
|
+
| {
|
|
60
|
+
kind: "command_passed"
|
|
61
|
+
/** The exact command the result claims passed (re-run verbatim). */
|
|
62
|
+
command: string
|
|
63
|
+
/** Optional output marker the re-run must contain (e.g. "PASS"). */
|
|
64
|
+
expectedMarker?: string
|
|
65
|
+
}
|
|
66
|
+
| {
|
|
67
|
+
kind: "serial_marker"
|
|
68
|
+
/** The claimed ordered markers, e.g. ["E1000: 1", "E1000: 2"]. */
|
|
69
|
+
markers: string[]
|
|
70
|
+
}
|
|
71
|
+
| {
|
|
72
|
+
kind: "pr_url"
|
|
73
|
+
/** The claimed pull-request number. */
|
|
74
|
+
prNumber: string
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** One claim's verification outcome. */
|
|
78
|
+
export interface ClaimVerification {
|
|
79
|
+
claim: ExtractedClaim
|
|
80
|
+
/** True only when ground truth positively confirms the claim. */
|
|
81
|
+
verified: boolean
|
|
82
|
+
/** Human-readable evidence for the verdict (fed to the deferral nudge). */
|
|
83
|
+
detail: string
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Options for verifyClaims (the loop resolves these from session config). */
|
|
87
|
+
export interface ClaimVerificationOptions {
|
|
88
|
+
workspaceRoot: string
|
|
89
|
+
permissions: PermissionsConfig
|
|
90
|
+
/** Cap on a single re-verification command's runtime, seconds (default 60). */
|
|
91
|
+
commandTimeoutS?: number
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Conservative extraction of machine-checkable claims from a completion
|
|
96
|
+
* result. See the module doc for the "specific AND verifiable" rule — this
|
|
97
|
+
* deliberately does NOT extract bare `make <target>` mentions, bare file
|
|
98
|
+
* paths, or bare numbers; each requires an affirmative claim verb nearby.
|
|
99
|
+
*/
|
|
100
|
+
export function extractClaims(result: string): ExtractedClaim[] {
|
|
101
|
+
const claims: ExtractedClaim[] = []
|
|
102
|
+
const seen = new Set<string>()
|
|
103
|
+
|
|
104
|
+
const add = (c: ExtractedClaim, key: string) => {
|
|
105
|
+
if (!seen.has(key)) {
|
|
106
|
+
seen.add(key)
|
|
107
|
+
claims.push(c)
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// --- File-existence claims ---------------------------------------------
|
|
112
|
+
// "wrote/created/added/updated/edited/implemented kernel/foo.curlee" —
|
|
113
|
+
// a concrete source-ish file path + an affirmative creation/change verb.
|
|
114
|
+
// A bounded gap between the verb and the path is allowed ("added a Makefile
|
|
115
|
+
// target scripts/run-e1000.sh" — 3 filler words), but the path itself must
|
|
116
|
+
// still be a real workspace-relative source-like file.
|
|
117
|
+
const fileVerb =
|
|
118
|
+
/\b(?:wrote|created|added|updated|edited|implemented|landed|merged|writes?|creates?|updates?|edits?)\b(?:\s+[A-Za-z0-9_-]+){0,4}\s+([A-Za-z0-9_./-]+\.(?:curlee|ts|js|py|sh|c|h|rs|go|json|md|mk))\b/gi
|
|
119
|
+
for (const m of result.matchAll(fileVerb)) {
|
|
120
|
+
const raw = m[1] as string
|
|
121
|
+
// Only workspace-relative-looking paths (never absolute /tmp or $HOME).
|
|
122
|
+
if (raw.startsWith("/") || raw.includes("..")) {
|
|
123
|
+
continue
|
|
124
|
+
}
|
|
125
|
+
// Skip paths that are clearly not files the model wrote (docs dirs are
|
|
126
|
+
// legitimately written too, so no filter there — existence is the check).
|
|
127
|
+
add({ kind: "file_exists", path: raw }, `file:${raw}`)
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// The fileVerb pattern above only sees a path that sits within ~4 word
|
|
131
|
+
// tokens of the verb. The other common real shape puts the changed
|
|
132
|
+
// artifact BEHIND a preposition, with arbitrary (often non-word:
|
|
133
|
+
// braces, quotes, colons) content between — verified live in the
|
|
134
|
+
// followthrough sweep: *"I added {"id": 7, "name": "lint"} to the tasks
|
|
135
|
+
// array in tasks.json."* The path there ("tasks.json") is ~40 chars past
|
|
136
|
+
// the verb and separated by JSON punctuation the `\s+word` gap can't
|
|
137
|
+
// cross. Match an affirmative change verb, then up to ~90 chars of any
|
|
138
|
+
// non-sentence-ending text, then an `in|into|to <path>` — the file the
|
|
139
|
+
// model says it changed. Existence is still the only check (a claim
|
|
140
|
+
// "added … to tasks.json" against a tasks.json that never got the edit
|
|
141
|
+
// isn't caught here when the file already existed for other reasons —
|
|
142
|
+
// that specific hole is closed loop-side by the re-read-only completion
|
|
143
|
+
// guard — but a claim naming a file that does not exist AT ALL is).
|
|
144
|
+
const fileVerbPrep =
|
|
145
|
+
/\b(?:added|inserted|appended|wrote|written|created|placed|moved|copied|saved)\b[^.\n]{0,90}?\b(?:into|onto|in|to)\s+([A-Za-z0-9_./-]+\.(?:curlee|ts|js|py|sh|c|h|rs|go|json|md|mk|txt|toml|ya?ml|cfg|ini))\b/gi
|
|
146
|
+
for (const m of result.matchAll(fileVerbPrep)) {
|
|
147
|
+
const raw = m[1] as string
|
|
148
|
+
if (raw.startsWith("/") || raw.includes("..")) {
|
|
149
|
+
continue
|
|
150
|
+
}
|
|
151
|
+
add({ kind: "file_exists", path: raw }, `file:${raw}`)
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// --- Command-passed claims ---------------------------------------------
|
|
155
|
+
// "make qemu-e1000-smoke passed" / "curlee check kernel/foo.curlee
|
|
156
|
+
// passed cleanly" / "npm test passed" — the verification-shaped commands
|
|
157
|
+
// this harness cares about, followed (within a bounded window) by an
|
|
158
|
+
// affirmative pass verb. This is the exact class of claim the e1000
|
|
159
|
+
// fabrication made ("all three hard gates pass").
|
|
160
|
+
//
|
|
161
|
+
// The pass-phrase group (m[3], the whole match text between the command
|
|
162
|
+
// and the pass verb — e.g. "cleanly", " with all checks", " (PASS)") is
|
|
163
|
+
// also scanned for a concrete output marker the re-run must CONTAIN, not
|
|
164
|
+
// just exit 0 on. A claim like "make check passed cleanly" names no
|
|
165
|
+
// specific marker and verifies on exit 0 alone (the marker field stays
|
|
166
|
+
// unset); a claim like "make check passed with 'PASS'" (or an uppercase
|
|
167
|
+
// word like "PASS"/"OK" in the pass phrase) is only verified when the
|
|
168
|
+
// re-ran command's real output actually contains that token. This closes
|
|
169
|
+
// plan A2's half-implementation: exit 0 alone is no longer enough when
|
|
170
|
+
// the model named a specific expected output.
|
|
171
|
+
const cmdClaim = /\b(make\s+[A-Za-z0-9_./-]+|npm\s+(?:test|run\s+[A-Za-z0-9_-]+)|curlee\s+(?:check|run)\s+[A-Za-z0-9_./-]+)\b([^.\n]{0,80}?)\b(pass(?:ed|es)?|clean(?:ly)?|green|succeed(?:ed)?|success)\b/gi
|
|
172
|
+
for (const m of result.matchAll(cmdClaim)) {
|
|
173
|
+
const command = (m[1] as string).trim()
|
|
174
|
+
const between = m[2] as string
|
|
175
|
+
const passPhrase = m[3] as string
|
|
176
|
+
// A specific output marker the re-run's output must contain. The
|
|
177
|
+
// marker usually sits AFTER the pass verb ("passed with 'PASS'",
|
|
178
|
+
// "passed: all tests PASS"), so look at a bounded window of the ORIGINAL
|
|
179
|
+
// result text following this match (sliced from the source, not
|
|
180
|
+
// consumed by matchAll — later claims must still match independently),
|
|
181
|
+
// cut at the sentence boundary. Prefer a quoted token ("passed 'PASS'",
|
|
182
|
+
// "output shows 'OK'") anywhere in the claim; fall back to an all-caps
|
|
183
|
+
// word ONLY in the text AFTER the pass verb ("passed: all tests PASS",
|
|
184
|
+
// "passed CLEANLY"), which is the conventional shape of a real gate's
|
|
185
|
+
// printed marker. The all-caps fallback deliberately never scans the
|
|
186
|
+
// `between` gap or the pass verb itself — "make check in CI PASSED" or
|
|
187
|
+
// an emphasized "PASSED" verb must not become an output marker.
|
|
188
|
+
// Lowercase adjectives like "cleanly" / "green" are NOT output markers
|
|
189
|
+
// either — they describe the pass, they don't name a token.
|
|
190
|
+
const afterStart = (m.index ?? 0) + m[0].length
|
|
191
|
+
const after = (result.slice(afterStart, afterStart + 60).split(/[.\n]/)[0] ?? "").trim()
|
|
192
|
+
const quoted = /['"]([A-Za-z0-9][A-Za-z0-9 _.:/-]{0,40})['"]/.exec(`${between} ${passPhrase} ${after}`)
|
|
193
|
+
const capped = /([A-Z]{2,})/.exec(after)
|
|
194
|
+
const expectedMarker = quoted?.[1] ?? capped?.[1]
|
|
195
|
+
add(
|
|
196
|
+
{
|
|
197
|
+
kind: "command_passed",
|
|
198
|
+
command,
|
|
199
|
+
...(expectedMarker ? { expectedMarker } : {}),
|
|
200
|
+
},
|
|
201
|
+
`cmd:${command}`,
|
|
202
|
+
)
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// --- Serial-marker claims ----------------------------------------------
|
|
206
|
+
// The known smoke-gate marker families ("E1000: 1", "NET: 3", "JSON: 1",
|
|
207
|
+
// "FB: 1", ...) when the result ALSO talks about serial/log/verification —
|
|
208
|
+
// the fabricated e1000 report claimed "E1000: 1 / E1000: 2" in the serial
|
|
209
|
+
// log. Extraction requires the marker family AND a nearby serial/log
|
|
210
|
+
// reference so unrelated numbers never get extracted.
|
|
211
|
+
const markerFamilies = /(E1000|NET|JSON|LLM|ARP|TCP|SND|RCV|TOOL|RX|FB|FR|RING):\s*\d+/g
|
|
212
|
+
const mentionsSerial = /serial|log|boot|qemu/i.test(result)
|
|
213
|
+
if (mentionsSerial) {
|
|
214
|
+
const markers = [...result.matchAll(markerFamilies)].map((m) => m[0].trim())
|
|
215
|
+
if (markers.length > 0) {
|
|
216
|
+
add({ kind: "serial_marker", markers }, `markers:${markers.join("|")}`)
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// --- PR-URL claims -----------------------------------------------------
|
|
221
|
+
// Any pull-request URL/number in a completion result is only trustworthy
|
|
222
|
+
// if real git history shows it — a fabricated "PR #123" was a documented
|
|
223
|
+
// incident shape (add_no_fabricated_report_examples.py). Extract and let
|
|
224
|
+
// verification fail-closed.
|
|
225
|
+
const prClaim = /pull\/(\d+)|PR\s*#?(\d+)/gi
|
|
226
|
+
for (const m of result.matchAll(prClaim)) {
|
|
227
|
+
const n = (m[1] ?? m[2]) as string
|
|
228
|
+
if (n) {
|
|
229
|
+
add({ kind: "pr_url", prNumber: n }, `pr:${n}`)
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
return claims
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* Path-safety check for a file claim: the resolved target must stay inside
|
|
238
|
+
* the workspace root (mirrors the executor's resolveWithinWorkspace posture).
|
|
239
|
+
*/
|
|
240
|
+
function resolveClaimPath(workspaceRoot: string, rel: string): string | null {
|
|
241
|
+
// Claims are workspace-relative ONLY — an absolute path is not a valid
|
|
242
|
+
// claim path (extractClaims already filters `/`-prefixed and `..` paths
|
|
243
|
+
// before they reach verification; this is defense in depth so a direct
|
|
244
|
+
// caller can never resolve an absolute path into a "verified" file).
|
|
245
|
+
if (path.isAbsolute(rel)) {
|
|
246
|
+
return null
|
|
247
|
+
}
|
|
248
|
+
const resolved = path.resolve(workspaceRoot, rel)
|
|
249
|
+
const root = path.resolve(workspaceRoot)
|
|
250
|
+
if (resolved !== root && !resolved.startsWith(root + path.sep)) {
|
|
251
|
+
return null
|
|
252
|
+
}
|
|
253
|
+
return resolved
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* Re-run a claimed command through the SAME permission gate as
|
|
258
|
+
* execute_command (deny-list, redirect-escape) with a bounded timeout.
|
|
259
|
+
* Returns { exitCode, output } — never throws for a non-zero exit (that's a
|
|
260
|
+
* failed verification, not a harness crash).
|
|
261
|
+
*/
|
|
262
|
+
async function rerunCommand(
|
|
263
|
+
command: string,
|
|
264
|
+
opts: ClaimVerificationOptions,
|
|
265
|
+
): Promise<{ exitCode: number | null; output: string; refused: string | null }> {
|
|
266
|
+
const timeoutS = opts.commandTimeoutS ?? 60
|
|
267
|
+
// Permission gate — same checks executeCommandHandler applies (executor.ts).
|
|
268
|
+
const redirectEscape = checkRedirectEscape(command, opts.workspaceRoot)
|
|
269
|
+
if (redirectEscape !== null) {
|
|
270
|
+
return { exitCode: null, output: "", refused: `redirect escapes workspace (${redirectEscape.operator} ${redirectEscape.word})` }
|
|
271
|
+
}
|
|
272
|
+
const refusal = checkCommand(command, opts.permissions.allowedCommands, opts.permissions.deniedCommands, {
|
|
273
|
+
workspaceRoot: opts.workspaceRoot,
|
|
274
|
+
})
|
|
275
|
+
if (refusal !== null) {
|
|
276
|
+
const detail =
|
|
277
|
+
refusal.kind === "denied"
|
|
278
|
+
? `denied (matched '${refusal.pattern ?? "?"}')`
|
|
279
|
+
: refusal.kind === "redirect_escape" && refusal.redirect
|
|
280
|
+
? `redirect escapes workspace (${refusal.redirect.operator} ${refusal.redirect.word})`
|
|
281
|
+
: refusal.kind === "not_allowed"
|
|
282
|
+
? "not on the allow-list"
|
|
283
|
+
: `denied (${refusal.kind})`
|
|
284
|
+
return { exitCode: null, output: "", refused: detail }
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
return new Promise((resolve) => {
|
|
288
|
+
let stdout = ""
|
|
289
|
+
let stderr = ""
|
|
290
|
+
let settled = false
|
|
291
|
+
const finish = (exitCode: number | null, output: string) => {
|
|
292
|
+
if (!settled) {
|
|
293
|
+
settled = true
|
|
294
|
+
resolve({ exitCode, output, refused: null })
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
const child = execFile(
|
|
298
|
+
BASH_PATH ?? "/bin/sh",
|
|
299
|
+
["-c", command],
|
|
300
|
+
{ cwd: opts.workspaceRoot, timeout: timeoutS * 1000, maxBuffer: 8 * 1024 * 1024 },
|
|
301
|
+
(err, so, se) => {
|
|
302
|
+
stdout = so
|
|
303
|
+
stderr = se
|
|
304
|
+
const code = err && typeof err.code === "number" ? err.code : err ? null : 0
|
|
305
|
+
finish(code, [stdout, stderr].filter(Boolean).join("\n"))
|
|
306
|
+
},
|
|
307
|
+
)
|
|
308
|
+
// A timeout kills the child (unlike execute_command's backgrounding) —
|
|
309
|
+
// verification re-runs are exactly the deterministic, bounded gates the
|
|
310
|
+
// session claims to have run; there is no legitimate "background" here.
|
|
311
|
+
child.on("error", () => finish(null, "spawn error during claim re-verification"))
|
|
312
|
+
})
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/**
|
|
316
|
+
* Find the newest build/serial-*.log (or any *.log under build/) for
|
|
317
|
+
* serial-marker claims. Mirrors the smoke-gate scripts' serial capture path.
|
|
318
|
+
*/
|
|
319
|
+
async function newestSerialLog(workspaceRoot: string): Promise<string | null> {
|
|
320
|
+
const buildDir = path.join(workspaceRoot, "build")
|
|
321
|
+
try {
|
|
322
|
+
const entries = await fsp.readdir(buildDir)
|
|
323
|
+
const logs = entries.filter((f) => /^serial-.*\.log$/.test(f) || /^serial.*\.log$/.test(f))
|
|
324
|
+
if (logs.length === 0) {
|
|
325
|
+
return null
|
|
326
|
+
}
|
|
327
|
+
// Newest by mtime.
|
|
328
|
+
let best: { name: string; mtime: number } | null = null
|
|
329
|
+
for (const name of logs) {
|
|
330
|
+
try {
|
|
331
|
+
const st = await fsp.stat(path.join(buildDir, name))
|
|
332
|
+
if (!best || st.mtimeMs > best.mtime) {
|
|
333
|
+
best = { name, mtime: st.mtimeMs }
|
|
334
|
+
}
|
|
335
|
+
} catch {
|
|
336
|
+
// unreadable entry — skip
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
return best ? path.join(buildDir, best.name) : null
|
|
340
|
+
} catch {
|
|
341
|
+
return null
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/** Verify one claim against ground truth. */
|
|
346
|
+
async function verifyClaim(
|
|
347
|
+
claim: ExtractedClaim,
|
|
348
|
+
opts: ClaimVerificationOptions,
|
|
349
|
+
): Promise<ClaimVerification> {
|
|
350
|
+
switch (claim.kind) {
|
|
351
|
+
case "file_exists": {
|
|
352
|
+
const resolved = resolveClaimPath(opts.workspaceRoot, claim.path)
|
|
353
|
+
if (!resolved) {
|
|
354
|
+
return {
|
|
355
|
+
claim,
|
|
356
|
+
verified: false,
|
|
357
|
+
detail: `claim path '${claim.path}' escapes the workspace`,
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
try {
|
|
361
|
+
const st = await fsp.stat(resolved)
|
|
362
|
+
if (st.size === 0) {
|
|
363
|
+
return { claim, verified: false, detail: `'${claim.path}' exists but is empty (0 bytes)` }
|
|
364
|
+
}
|
|
365
|
+
return { claim, verified: true, detail: `'${claim.path}' exists on disk (${st.size} bytes)` }
|
|
366
|
+
} catch {
|
|
367
|
+
return { claim, verified: false, detail: `no file '${claim.path}' exists on disk` }
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
case "command_passed": {
|
|
372
|
+
const { exitCode, output, refused } = await rerunCommand(claim.command, opts)
|
|
373
|
+
if (refused) {
|
|
374
|
+
return { claim, verified: false, detail: `re-verification refused by permission gate: ${refused}` }
|
|
375
|
+
}
|
|
376
|
+
if (exitCode !== 0) {
|
|
377
|
+
return {
|
|
378
|
+
claim,
|
|
379
|
+
verified: false,
|
|
380
|
+
detail: `re-ran '${claim.command}' → exit code ${exitCode ?? "unknown"}; the claimed pass is not confirmed`,
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
if (claim.expectedMarker && !output.includes(claim.expectedMarker)) {
|
|
384
|
+
return {
|
|
385
|
+
claim,
|
|
386
|
+
verified: false,
|
|
387
|
+
detail: `re-ran '${claim.command}' → exit 0 but output lacks expected marker '${claim.expectedMarker}'`,
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
return { claim, verified: true, detail: `re-ran '${claim.command}' → exit 0` }
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
case "serial_marker": {
|
|
394
|
+
const logPath = await newestSerialLog(opts.workspaceRoot)
|
|
395
|
+
if (!logPath) {
|
|
396
|
+
return { claim, verified: false, detail: `no build/serial-*.log exists to confirm markers ${claim.markers.join(", ")}` }
|
|
397
|
+
}
|
|
398
|
+
try {
|
|
399
|
+
const content = await fsp.readFile(logPath, "utf-8")
|
|
400
|
+
const missing = claim.markers.filter((m) => !content.includes(m))
|
|
401
|
+
if (missing.length > 0) {
|
|
402
|
+
return {
|
|
403
|
+
claim,
|
|
404
|
+
verified: false,
|
|
405
|
+
detail: `serial log ${path.basename(logPath)} lacks marker(s): ${missing.join(", ")}`,
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
// Ordered check: each marker's index must be >= the previous one's.
|
|
409
|
+
let last = -1
|
|
410
|
+
for (const m of claim.markers) {
|
|
411
|
+
const idx = content.indexOf(m)
|
|
412
|
+
if (idx < last) {
|
|
413
|
+
return {
|
|
414
|
+
claim,
|
|
415
|
+
verified: false,
|
|
416
|
+
detail: `serial log ${path.basename(logPath)} has markers but not in claimed order (${m} precedes an earlier marker)`,
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
last = idx
|
|
420
|
+
}
|
|
421
|
+
return { claim, verified: true, detail: `serial log ${path.basename(logPath)} contains all claimed markers in order` }
|
|
422
|
+
} catch {
|
|
423
|
+
return { claim, verified: false, detail: `could not read serial log ${logPath}` }
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
case "pr_url": {
|
|
428
|
+
// A PR number is only verified by real git/gh evidence: search the
|
|
429
|
+
// last 50 commit subjects for the number. No gh call (avoid a
|
|
430
|
+
// network dependency in a verification path) — git history is the
|
|
431
|
+
// deterministic, offline ground truth for "a PR with this number
|
|
432
|
+
// was actually involved".
|
|
433
|
+
return new Promise((resolve) => {
|
|
434
|
+
execFile(
|
|
435
|
+
"git",
|
|
436
|
+
["-C", opts.workspaceRoot, "log", "--oneline", "-50"],
|
|
437
|
+
{ timeout: 10_000, maxBuffer: 1024 * 1024 },
|
|
438
|
+
(err, stdout) => {
|
|
439
|
+
if (err) {
|
|
440
|
+
resolve({ claim, verified: false, detail: "no git history available to confirm PR evidence" })
|
|
441
|
+
return
|
|
442
|
+
}
|
|
443
|
+
if (stdout.includes(`#${claim.prNumber}`) || stdout.includes(`pull/${claim.prNumber}`)) {
|
|
444
|
+
resolve({ claim, verified: true, detail: `git history references PR #${claim.prNumber}` })
|
|
445
|
+
} else {
|
|
446
|
+
resolve({
|
|
447
|
+
claim,
|
|
448
|
+
verified: false,
|
|
449
|
+
detail: `no git history entry references PR #${claim.prNumber} — no real PR side effect found`,
|
|
450
|
+
})
|
|
451
|
+
}
|
|
452
|
+
},
|
|
453
|
+
)
|
|
454
|
+
})
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
/**
|
|
460
|
+
* Verify a list of extracted claims against ground truth. Returns the full
|
|
461
|
+
* per-claim results; the caller (the loop) treats ANY unverified claim as
|
|
462
|
+
* grounds to defer the completion.
|
|
463
|
+
*/
|
|
464
|
+
export async function verifyClaims(
|
|
465
|
+
claims: ExtractedClaim[],
|
|
466
|
+
opts: ClaimVerificationOptions,
|
|
467
|
+
): Promise<ClaimVerification[]> {
|
|
468
|
+
const results: ClaimVerification[] = []
|
|
469
|
+
for (const claim of claims) {
|
|
470
|
+
// Sequential on purpose: verification re-runs commands that share the
|
|
471
|
+
// host's shell; parallel re-runs could interfere (and the count is
|
|
472
|
+
// small — a completion report has at most a handful of claims).
|
|
473
|
+
results.push(await verifyClaim(claim, opts))
|
|
474
|
+
}
|
|
475
|
+
return results
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
/** Convenience: a stable human-readable label for a claim (for deferral messages). */
|
|
479
|
+
export function claimLabel(claim: ExtractedClaim): string {
|
|
480
|
+
switch (claim.kind) {
|
|
481
|
+
case "file_exists":
|
|
482
|
+
return `the file '${claim.path}' exists`
|
|
483
|
+
case "command_passed":
|
|
484
|
+
return `the command '${claim.command}' passed`
|
|
485
|
+
case "serial_marker":
|
|
486
|
+
return `the serial markers ${claim.markers.join(", ")} appear in order in a serial log`
|
|
487
|
+
case "pr_url":
|
|
488
|
+
return `a real PR #${claim.prNumber} exists`
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
/** Convenience: did a claim set pass verification entirely? */
|
|
493
|
+
export function allClaimsVerified(results: ClaimVerification[]): boolean {
|
|
494
|
+
return results.length > 0 && results.every((r) => r.verified)
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
/** Convenience: the first unverified claim's detail (for the deferral nudge). */
|
|
498
|
+
export function firstUnverifiedDetail(results: ClaimVerification[]): string | undefined {
|
|
499
|
+
return results.find((r) => !r.verified)?.detail
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
/** Re-exported for tests: the workspace-path resolver (kept internal otherwise). */
|
|
503
|
+
export { resolveClaimPath }
|
package/src/engine/events.ts
CHANGED
|
@@ -14,6 +14,11 @@
|
|
|
14
14
|
* session_start / iteration_start / llm_response / tool_call / tool_result
|
|
15
15
|
* checkpoint_saved / decision_blocked / decision_answered
|
|
16
16
|
* condensed / todo_updated / paused / resumed / session_end
|
|
17
|
+
* unverified_claim (fabrication fix, 2026-09-01): emitted when
|
|
18
|
+
* evidenceRequiredCompletion was on and an attempt_completion was DEFERRED
|
|
19
|
+
* because a machine-checkable claim in its result could not be verified
|
|
20
|
+
* against ground truth — carries the verification counts + the first
|
|
21
|
+
* unverified claim's detail (see EventFeed.unverifiedClaim)
|
|
17
22
|
* attempt_completion (issue #34): the session's FINAL report, emitted when
|
|
18
23
|
* the loop accepts completion (attempt_completion tool call OR the
|
|
19
24
|
* text-only success fallback) and carries the FULL report text — this is
|
|
@@ -300,6 +305,33 @@ export class EventFeed {
|
|
|
300
305
|
})
|
|
301
306
|
}
|
|
302
307
|
|
|
308
|
+
/**
|
|
309
|
+
* Evidence-gated completion (fabrication fix, 2026-09-01): emitted when
|
|
310
|
+
* evidenceRequiredCompletion was on and an attempt_completion was DEFERRED
|
|
311
|
+
* because one or more machine-checkable claims in its result (a file
|
|
312
|
+
* exists, a command passed, serial markers appear, a PR exists) could not
|
|
313
|
+
* be independently verified against ground truth (src/engine/claims.ts).
|
|
314
|
+
* Carries the verification counts + the first unverified claim's detail so
|
|
315
|
+
* downstream consumers (eval, selfplay miner, orchestrator) can see WHY a
|
|
316
|
+
* completion was refused — the structured record that a fabrication
|
|
317
|
+
* attempt was caught, not just a log line.
|
|
318
|
+
*/
|
|
319
|
+
unverifiedClaim(fields: {
|
|
320
|
+
iteration: number
|
|
321
|
+
claimsChecked: number
|
|
322
|
+
claimsPassed: number
|
|
323
|
+
claimsUnverified: number
|
|
324
|
+
detail: string
|
|
325
|
+
}): Promise<void> {
|
|
326
|
+
return this.emit("unverified_claim", {
|
|
327
|
+
iteration: fields.iteration,
|
|
328
|
+
claimsChecked: fields.claimsChecked,
|
|
329
|
+
claimsPassed: fields.claimsPassed,
|
|
330
|
+
claimsUnverified: fields.claimsUnverified,
|
|
331
|
+
detail: truncateField(fields.detail).text,
|
|
332
|
+
})
|
|
333
|
+
}
|
|
334
|
+
|
|
303
335
|
/** iteration 0 = the session's baseline checkpoint (before iteration 1). */
|
|
304
336
|
checkpointSaved(iteration: number): Promise<void> {
|
|
305
337
|
return this.emit("checkpoint_saved", { iteration })
|
package/src/engine/lazy-tools.ts
CHANGED
|
@@ -49,7 +49,6 @@ export const CORE_TOOL_NAMES = new Set([
|
|
|
49
49
|
"attempt_completion",
|
|
50
50
|
"execute_command",
|
|
51
51
|
"list_files",
|
|
52
|
-
"new_task",
|
|
53
52
|
"read_file",
|
|
54
53
|
// search_replace deliberately excluded — same reasoning as apply_diff
|
|
55
54
|
// above it in git history: search_replace requires an EXACT literal
|
|
@@ -74,9 +73,23 @@ export const CORE_TOOL_NAMES = new Set([
|
|
|
74
73
|
// call it correctly but only as narrated text, never having pulled its
|
|
75
74
|
// real schema in via request_tool first.
|
|
76
75
|
"set_indentation",
|
|
77
|
-
"switch_mode",
|
|
78
|
-
"update_todo_list",
|
|
79
76
|
"write_to_file",
|
|
77
|
+
// 2026-09-02: new_task, switch_mode, and update_todo_list demoted from
|
|
78
|
+
// core to lazy — real usage data across 62 logged local-backend eval
|
|
79
|
+
// sessions (grep "tool result: <name>" across eval_verifier_runs/*/
|
|
80
|
+
// session.log) showed ZERO calls to any of these three, ever, while
|
|
81
|
+
// still paying ~3,700 combined chars of their schemas on every single
|
|
82
|
+
// turn of every session. Unlike set_indentation (which has a specific
|
|
83
|
+
// documented live failure showing the model won't request_tool it when
|
|
84
|
+
// it's actually needed), there is no equivalent evidence for these
|
|
85
|
+
// three — no session in that corpus needed delegation (new_task), a
|
|
86
|
+
// mode switch (switch_mode), or a multi-step plan register
|
|
87
|
+
// (update_todo_list) at all, since these are single-file scratch
|
|
88
|
+
// eval tasks. If a session genuinely needs one, it's still one
|
|
89
|
+
// request_tool call away. Revisit if live data ever shows a session
|
|
90
|
+
// that needed one of these three but never called request_tool for it
|
|
91
|
+
// (the set_indentation failure shape) — that would argue for
|
|
92
|
+
// re-promoting that specific tool back to core.
|
|
80
93
|
])
|
|
81
94
|
|
|
82
95
|
export const LIST_TOOLS_NAME = "list_tools"
|
|
@@ -111,17 +124,25 @@ export function splitCoreAndLazyTools(allTools: ChatTool[]): SplitTools {
|
|
|
111
124
|
}
|
|
112
125
|
|
|
113
126
|
export function buildListToolsTool(): ChatTool {
|
|
127
|
+
// 2026-09-02: built from CORE_TOOL_NAMES itself rather than a hand-
|
|
128
|
+
// written duplicate list — the previous static string had already
|
|
129
|
+
// drifted (it named apply_diff/search_replace as "always available",
|
|
130
|
+
// which was never true; both are deliberately lazy, see
|
|
131
|
+
// CORE_TOOL_NAMES's own comments) and would have drifted again the
|
|
132
|
+
// moment core membership changed without this description changing
|
|
133
|
+
// with it. request_tool/list_tools themselves are the delivery
|
|
134
|
+
// mechanism, not part of the "core" the model chooses among, so they're
|
|
135
|
+
// deliberately left out of this parenthetical (the tool's own name
|
|
136
|
+
// already makes clear it exists).
|
|
137
|
+
const coreList = [...CORE_TOOL_NAMES].join(", ")
|
|
114
138
|
return {
|
|
115
139
|
type: "function",
|
|
116
140
|
function: {
|
|
117
141
|
name: LIST_TOOLS_NAME,
|
|
118
142
|
description:
|
|
119
|
-
|
|
120
|
-
"
|
|
121
|
-
"
|
|
122
|
-
"switch_mode, new_task, update_todo_list — always available, not listed " +
|
|
123
|
-
"here). Call request_tool with a name from this list to make that tool " +
|
|
124
|
-
"callable on your NEXT turn.",
|
|
143
|
+
`List additional tools available in this session beyond the core set (${coreList} — ` +
|
|
144
|
+
"always available, not listed here). Call request_tool with a name from this list " +
|
|
145
|
+
"to make that tool callable on your NEXT turn.",
|
|
125
146
|
parameters: { type: "object", properties: {}, required: [] },
|
|
126
147
|
},
|
|
127
148
|
}
|