headlesscode 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,503 @@
1
+ /**
2
+ * Evidence-gated completion — the fabrication-fix core (2026-09-01).
3
+ *
4
+ * The FINAL_REPORT's central finding (joeos_finetune_data/FINAL_REPORT.md,
5
+ * §4): a real headlesscode session reported complete success with a
6
+ * fabricated QEMU serial log excerpt ("E1000: 1 / E1000: 2") claiming three
7
+ * hard gates passed, when the driver file was never merged and the claimed
8
+ * `make qemu-e1000-smoke` target did not even exist. Every pre-existing
9
+ * completion guardrail was token-based — it checked *wording* (did the last
10
+ * command fail? did any write tool ever get called?), never *ground truth*
11
+ * (does the file the model names exist with the content implied? did the
12
+ * exact command the model names actually run and pass?).
13
+ *
14
+ * This module closes that gap with the same philosophy the orchestrator's
15
+ * verification-gate.ts already applies to review/QA verdicts: the model's
16
+ * own prose is never a source of truth about what really happened. Ground
17
+ * truth comes from the filesystem and real re-runs, never from the report.
18
+ *
19
+ * Two stages:
20
+ *
21
+ * 1. `extractClaims` — conservatively parse an attempt_completion result
22
+ * for machine-checkable claims. ONLY claims that are BOTH specific
23
+ * (name a concrete file, command, marker, or PR) AND verifiable get
24
+ * extracted. Vague prose ("the driver works") is not a checkable claim
25
+ * and never gates — but it also can never *pass* a gate.
26
+ * 2. `verifyClaims` — check each extracted claim against ground truth:
27
+ * file_exists → fs.stat on the resolved path (path-safety enforced)
28
+ * command_passed→ RE-RUN the exact command through the same
29
+ * permission gate as execute_command, require exit 0
30
+ * (+ optional expected output marker)
31
+ * serial_marker → grep the newest build/serial-*.log for the claimed
32
+ * ordered markers (mirrors the qemu-smoke gate idiom)
33
+ * pr_url → only verified if the PR number appears in real
34
+ * git history (a real gh/git side effect); else
35
+ * fail-closed
36
+ *
37
+ * Fail-closed posture throughout: a claim that cannot be verified (missing
38
+ * file, failed re-run, no serial log, no git evidence) is UNVERIFIED, never
39
+ * "probably fine". The loop refuses the completion and names the specific
40
+ * unverified claim so the model has something concrete to fix.
41
+ */
42
+
43
+ import * as fs from "node:fs"
44
+ import * as fsp from "node:fs/promises"
45
+ import * as path from "node:path"
46
+ import { execFile } from "node:child_process"
47
+
48
+ import { checkCommand, checkRedirectEscape } from "../permissions/commands.js"
49
+ import type { PermissionsConfig } from "../permissions/config.js"
50
+ import { BASH_PATH } from "../tools/executor.js"
51
+
52
+ /** A machine-checkable claim extracted from an attempt_completion result. */
53
+ export type ExtractedClaim =
54
+ | {
55
+ kind: "file_exists"
56
+ /** The workspace-relative path the result claims exists (posix). */
57
+ path: string
58
+ }
59
+ | {
60
+ kind: "command_passed"
61
+ /** The exact command the result claims passed (re-run verbatim). */
62
+ command: string
63
+ /** Optional output marker the re-run must contain (e.g. "PASS"). */
64
+ expectedMarker?: string
65
+ }
66
+ | {
67
+ kind: "serial_marker"
68
+ /** The claimed ordered markers, e.g. ["E1000: 1", "E1000: 2"]. */
69
+ markers: string[]
70
+ }
71
+ | {
72
+ kind: "pr_url"
73
+ /** The claimed pull-request number. */
74
+ prNumber: string
75
+ }
76
+
77
+ /** One claim's verification outcome. */
78
+ export interface ClaimVerification {
79
+ claim: ExtractedClaim
80
+ /** True only when ground truth positively confirms the claim. */
81
+ verified: boolean
82
+ /** Human-readable evidence for the verdict (fed to the deferral nudge). */
83
+ detail: string
84
+ }
85
+
86
+ /** Options for verifyClaims (the loop resolves these from session config). */
87
+ export interface ClaimVerificationOptions {
88
+ workspaceRoot: string
89
+ permissions: PermissionsConfig
90
+ /** Cap on a single re-verification command's runtime, seconds (default 60). */
91
+ commandTimeoutS?: number
92
+ }
93
+
94
+ /**
95
+ * Conservative extraction of machine-checkable claims from a completion
96
+ * result. See the module doc for the "specific AND verifiable" rule — this
97
+ * deliberately does NOT extract bare `make <target>` mentions, bare file
98
+ * paths, or bare numbers; each requires an affirmative claim verb nearby.
99
+ */
100
+ export function extractClaims(result: string): ExtractedClaim[] {
101
+ const claims: ExtractedClaim[] = []
102
+ const seen = new Set<string>()
103
+
104
+ const add = (c: ExtractedClaim, key: string) => {
105
+ if (!seen.has(key)) {
106
+ seen.add(key)
107
+ claims.push(c)
108
+ }
109
+ }
110
+
111
+ // --- File-existence claims ---------------------------------------------
112
+ // "wrote/created/added/updated/edited/implemented kernel/foo.curlee" —
113
+ // a concrete source-ish file path + an affirmative creation/change verb.
114
+ // A bounded gap between the verb and the path is allowed ("added a Makefile
115
+ // target scripts/run-e1000.sh" — 3 filler words), but the path itself must
116
+ // still be a real workspace-relative source-like file.
117
+ const fileVerb =
118
+ /\b(?:wrote|created|added|updated|edited|implemented|landed|merged|writes?|creates?|updates?|edits?)\b(?:\s+[A-Za-z0-9_-]+){0,4}\s+([A-Za-z0-9_./-]+\.(?:curlee|ts|js|py|sh|c|h|rs|go|json|md|mk))\b/gi
119
+ for (const m of result.matchAll(fileVerb)) {
120
+ const raw = m[1] as string
121
+ // Only workspace-relative-looking paths (never absolute /tmp or $HOME).
122
+ if (raw.startsWith("/") || raw.includes("..")) {
123
+ continue
124
+ }
125
+ // Skip paths that are clearly not files the model wrote (docs dirs are
126
+ // legitimately written too, so no filter there — existence is the check).
127
+ add({ kind: "file_exists", path: raw }, `file:${raw}`)
128
+ }
129
+
130
+ // The fileVerb pattern above only sees a path that sits within ~4 word
131
+ // tokens of the verb. The other common real shape puts the changed
132
+ // artifact BEHIND a preposition, with arbitrary (often non-word:
133
+ // braces, quotes, colons) content between — verified live in the
134
+ // followthrough sweep: *"I added {"id": 7, "name": "lint"} to the tasks
135
+ // array in tasks.json."* The path there ("tasks.json") is ~40 chars past
136
+ // the verb and separated by JSON punctuation the `\s+word` gap can't
137
+ // cross. Match an affirmative change verb, then up to ~90 chars of any
138
+ // non-sentence-ending text, then an `in|into|to <path>` — the file the
139
+ // model says it changed. Existence is still the only check (a claim
140
+ // "added … to tasks.json" against a tasks.json that never got the edit
141
+ // isn't caught here when the file already existed for other reasons —
142
+ // that specific hole is closed loop-side by the re-read-only completion
143
+ // guard — but a claim naming a file that does not exist AT ALL is).
144
+ const fileVerbPrep =
145
+ /\b(?:added|inserted|appended|wrote|written|created|placed|moved|copied|saved)\b[^.\n]{0,90}?\b(?:into|onto|in|to)\s+([A-Za-z0-9_./-]+\.(?:curlee|ts|js|py|sh|c|h|rs|go|json|md|mk|txt|toml|ya?ml|cfg|ini))\b/gi
146
+ for (const m of result.matchAll(fileVerbPrep)) {
147
+ const raw = m[1] as string
148
+ if (raw.startsWith("/") || raw.includes("..")) {
149
+ continue
150
+ }
151
+ add({ kind: "file_exists", path: raw }, `file:${raw}`)
152
+ }
153
+
154
+ // --- Command-passed claims ---------------------------------------------
155
+ // "make qemu-e1000-smoke passed" / "curlee check kernel/foo.curlee
156
+ // passed cleanly" / "npm test passed" — the verification-shaped commands
157
+ // this harness cares about, followed (within a bounded window) by an
158
+ // affirmative pass verb. This is the exact class of claim the e1000
159
+ // fabrication made ("all three hard gates pass").
160
+ //
161
+ // The pass-phrase group (m[3], the whole match text between the command
162
+ // and the pass verb — e.g. "cleanly", " with all checks", " (PASS)") is
163
+ // also scanned for a concrete output marker the re-run must CONTAIN, not
164
+ // just exit 0 on. A claim like "make check passed cleanly" names no
165
+ // specific marker and verifies on exit 0 alone (the marker field stays
166
+ // unset); a claim like "make check passed with 'PASS'" (or an uppercase
167
+ // word like "PASS"/"OK" in the pass phrase) is only verified when the
168
+ // re-ran command's real output actually contains that token. This closes
169
+ // plan A2's half-implementation: exit 0 alone is no longer enough when
170
+ // the model named a specific expected output.
171
+ const cmdClaim = /\b(make\s+[A-Za-z0-9_./-]+|npm\s+(?:test|run\s+[A-Za-z0-9_-]+)|curlee\s+(?:check|run)\s+[A-Za-z0-9_./-]+)\b([^.\n]{0,80}?)\b(pass(?:ed|es)?|clean(?:ly)?|green|succeed(?:ed)?|success)\b/gi
172
+ for (const m of result.matchAll(cmdClaim)) {
173
+ const command = (m[1] as string).trim()
174
+ const between = m[2] as string
175
+ const passPhrase = m[3] as string
176
+ // A specific output marker the re-run's output must contain. The
177
+ // marker usually sits AFTER the pass verb ("passed with 'PASS'",
178
+ // "passed: all tests PASS"), so look at a bounded window of the ORIGINAL
179
+ // result text following this match (sliced from the source, not
180
+ // consumed by matchAll — later claims must still match independently),
181
+ // cut at the sentence boundary. Prefer a quoted token ("passed 'PASS'",
182
+ // "output shows 'OK'") anywhere in the claim; fall back to an all-caps
183
+ // word ONLY in the text AFTER the pass verb ("passed: all tests PASS",
184
+ // "passed CLEANLY"), which is the conventional shape of a real gate's
185
+ // printed marker. The all-caps fallback deliberately never scans the
186
+ // `between` gap or the pass verb itself — "make check in CI PASSED" or
187
+ // an emphasized "PASSED" verb must not become an output marker.
188
+ // Lowercase adjectives like "cleanly" / "green" are NOT output markers
189
+ // either — they describe the pass, they don't name a token.
190
+ const afterStart = (m.index ?? 0) + m[0].length
191
+ const after = (result.slice(afterStart, afterStart + 60).split(/[.\n]/)[0] ?? "").trim()
192
+ const quoted = /['"]([A-Za-z0-9][A-Za-z0-9 _.:/-]{0,40})['"]/.exec(`${between} ${passPhrase} ${after}`)
193
+ const capped = /([A-Z]{2,})/.exec(after)
194
+ const expectedMarker = quoted?.[1] ?? capped?.[1]
195
+ add(
196
+ {
197
+ kind: "command_passed",
198
+ command,
199
+ ...(expectedMarker ? { expectedMarker } : {}),
200
+ },
201
+ `cmd:${command}`,
202
+ )
203
+ }
204
+
205
+ // --- Serial-marker claims ----------------------------------------------
206
+ // The known smoke-gate marker families ("E1000: 1", "NET: 3", "JSON: 1",
207
+ // "FB: 1", ...) when the result ALSO talks about serial/log/verification —
208
+ // the fabricated e1000 report claimed "E1000: 1 / E1000: 2" in the serial
209
+ // log. Extraction requires the marker family AND a nearby serial/log
210
+ // reference so unrelated numbers never get extracted.
211
+ const markerFamilies = /(E1000|NET|JSON|LLM|ARP|TCP|SND|RCV|TOOL|RX|FB|FR|RING):\s*\d+/g
212
+ const mentionsSerial = /serial|log|boot|qemu/i.test(result)
213
+ if (mentionsSerial) {
214
+ const markers = [...result.matchAll(markerFamilies)].map((m) => m[0].trim())
215
+ if (markers.length > 0) {
216
+ add({ kind: "serial_marker", markers }, `markers:${markers.join("|")}`)
217
+ }
218
+ }
219
+
220
+ // --- PR-URL claims -----------------------------------------------------
221
+ // Any pull-request URL/number in a completion result is only trustworthy
222
+ // if real git history shows it — a fabricated "PR #123" was a documented
223
+ // incident shape (add_no_fabricated_report_examples.py). Extract and let
224
+ // verification fail-closed.
225
+ const prClaim = /pull\/(\d+)|PR\s*#?(\d+)/gi
226
+ for (const m of result.matchAll(prClaim)) {
227
+ const n = (m[1] ?? m[2]) as string
228
+ if (n) {
229
+ add({ kind: "pr_url", prNumber: n }, `pr:${n}`)
230
+ }
231
+ }
232
+
233
+ return claims
234
+ }
235
+
236
+ /**
237
+ * Path-safety check for a file claim: the resolved target must stay inside
238
+ * the workspace root (mirrors the executor's resolveWithinWorkspace posture).
239
+ */
240
+ function resolveClaimPath(workspaceRoot: string, rel: string): string | null {
241
+ // Claims are workspace-relative ONLY — an absolute path is not a valid
242
+ // claim path (extractClaims already filters `/`-prefixed and `..` paths
243
+ // before they reach verification; this is defense in depth so a direct
244
+ // caller can never resolve an absolute path into a "verified" file).
245
+ if (path.isAbsolute(rel)) {
246
+ return null
247
+ }
248
+ const resolved = path.resolve(workspaceRoot, rel)
249
+ const root = path.resolve(workspaceRoot)
250
+ if (resolved !== root && !resolved.startsWith(root + path.sep)) {
251
+ return null
252
+ }
253
+ return resolved
254
+ }
255
+
256
+ /**
257
+ * Re-run a claimed command through the SAME permission gate as
258
+ * execute_command (deny-list, redirect-escape) with a bounded timeout.
259
+ * Returns { exitCode, output } — never throws for a non-zero exit (that's a
260
+ * failed verification, not a harness crash).
261
+ */
262
+ async function rerunCommand(
263
+ command: string,
264
+ opts: ClaimVerificationOptions,
265
+ ): Promise<{ exitCode: number | null; output: string; refused: string | null }> {
266
+ const timeoutS = opts.commandTimeoutS ?? 60
267
+ // Permission gate — same checks executeCommandHandler applies (executor.ts).
268
+ const redirectEscape = checkRedirectEscape(command, opts.workspaceRoot)
269
+ if (redirectEscape !== null) {
270
+ return { exitCode: null, output: "", refused: `redirect escapes workspace (${redirectEscape.operator} ${redirectEscape.word})` }
271
+ }
272
+ const refusal = checkCommand(command, opts.permissions.allowedCommands, opts.permissions.deniedCommands, {
273
+ workspaceRoot: opts.workspaceRoot,
274
+ })
275
+ if (refusal !== null) {
276
+ const detail =
277
+ refusal.kind === "denied"
278
+ ? `denied (matched '${refusal.pattern ?? "?"}')`
279
+ : refusal.kind === "redirect_escape" && refusal.redirect
280
+ ? `redirect escapes workspace (${refusal.redirect.operator} ${refusal.redirect.word})`
281
+ : refusal.kind === "not_allowed"
282
+ ? "not on the allow-list"
283
+ : `denied (${refusal.kind})`
284
+ return { exitCode: null, output: "", refused: detail }
285
+ }
286
+
287
+ return new Promise((resolve) => {
288
+ let stdout = ""
289
+ let stderr = ""
290
+ let settled = false
291
+ const finish = (exitCode: number | null, output: string) => {
292
+ if (!settled) {
293
+ settled = true
294
+ resolve({ exitCode, output, refused: null })
295
+ }
296
+ }
297
+ const child = execFile(
298
+ BASH_PATH ?? "/bin/sh",
299
+ ["-c", command],
300
+ { cwd: opts.workspaceRoot, timeout: timeoutS * 1000, maxBuffer: 8 * 1024 * 1024 },
301
+ (err, so, se) => {
302
+ stdout = so
303
+ stderr = se
304
+ const code = err && typeof err.code === "number" ? err.code : err ? null : 0
305
+ finish(code, [stdout, stderr].filter(Boolean).join("\n"))
306
+ },
307
+ )
308
+ // A timeout kills the child (unlike execute_command's backgrounding) —
309
+ // verification re-runs are exactly the deterministic, bounded gates the
310
+ // session claims to have run; there is no legitimate "background" here.
311
+ child.on("error", () => finish(null, "spawn error during claim re-verification"))
312
+ })
313
+ }
314
+
315
+ /**
316
+ * Find the newest build/serial-*.log (or any *.log under build/) for
317
+ * serial-marker claims. Mirrors the smoke-gate scripts' serial capture path.
318
+ */
319
+ async function newestSerialLog(workspaceRoot: string): Promise<string | null> {
320
+ const buildDir = path.join(workspaceRoot, "build")
321
+ try {
322
+ const entries = await fsp.readdir(buildDir)
323
+ const logs = entries.filter((f) => /^serial-.*\.log$/.test(f) || /^serial.*\.log$/.test(f))
324
+ if (logs.length === 0) {
325
+ return null
326
+ }
327
+ // Newest by mtime.
328
+ let best: { name: string; mtime: number } | null = null
329
+ for (const name of logs) {
330
+ try {
331
+ const st = await fsp.stat(path.join(buildDir, name))
332
+ if (!best || st.mtimeMs > best.mtime) {
333
+ best = { name, mtime: st.mtimeMs }
334
+ }
335
+ } catch {
336
+ // unreadable entry — skip
337
+ }
338
+ }
339
+ return best ? path.join(buildDir, best.name) : null
340
+ } catch {
341
+ return null
342
+ }
343
+ }
344
+
345
+ /** Verify one claim against ground truth. */
346
+ async function verifyClaim(
347
+ claim: ExtractedClaim,
348
+ opts: ClaimVerificationOptions,
349
+ ): Promise<ClaimVerification> {
350
+ switch (claim.kind) {
351
+ case "file_exists": {
352
+ const resolved = resolveClaimPath(opts.workspaceRoot, claim.path)
353
+ if (!resolved) {
354
+ return {
355
+ claim,
356
+ verified: false,
357
+ detail: `claim path '${claim.path}' escapes the workspace`,
358
+ }
359
+ }
360
+ try {
361
+ const st = await fsp.stat(resolved)
362
+ if (st.size === 0) {
363
+ return { claim, verified: false, detail: `'${claim.path}' exists but is empty (0 bytes)` }
364
+ }
365
+ return { claim, verified: true, detail: `'${claim.path}' exists on disk (${st.size} bytes)` }
366
+ } catch {
367
+ return { claim, verified: false, detail: `no file '${claim.path}' exists on disk` }
368
+ }
369
+ }
370
+
371
+ case "command_passed": {
372
+ const { exitCode, output, refused } = await rerunCommand(claim.command, opts)
373
+ if (refused) {
374
+ return { claim, verified: false, detail: `re-verification refused by permission gate: ${refused}` }
375
+ }
376
+ if (exitCode !== 0) {
377
+ return {
378
+ claim,
379
+ verified: false,
380
+ detail: `re-ran '${claim.command}' → exit code ${exitCode ?? "unknown"}; the claimed pass is not confirmed`,
381
+ }
382
+ }
383
+ if (claim.expectedMarker && !output.includes(claim.expectedMarker)) {
384
+ return {
385
+ claim,
386
+ verified: false,
387
+ detail: `re-ran '${claim.command}' → exit 0 but output lacks expected marker '${claim.expectedMarker}'`,
388
+ }
389
+ }
390
+ return { claim, verified: true, detail: `re-ran '${claim.command}' → exit 0` }
391
+ }
392
+
393
+ case "serial_marker": {
394
+ const logPath = await newestSerialLog(opts.workspaceRoot)
395
+ if (!logPath) {
396
+ return { claim, verified: false, detail: `no build/serial-*.log exists to confirm markers ${claim.markers.join(", ")}` }
397
+ }
398
+ try {
399
+ const content = await fsp.readFile(logPath, "utf-8")
400
+ const missing = claim.markers.filter((m) => !content.includes(m))
401
+ if (missing.length > 0) {
402
+ return {
403
+ claim,
404
+ verified: false,
405
+ detail: `serial log ${path.basename(logPath)} lacks marker(s): ${missing.join(", ")}`,
406
+ }
407
+ }
408
+ // Ordered check: each marker's index must be >= the previous one's.
409
+ let last = -1
410
+ for (const m of claim.markers) {
411
+ const idx = content.indexOf(m)
412
+ if (idx < last) {
413
+ return {
414
+ claim,
415
+ verified: false,
416
+ detail: `serial log ${path.basename(logPath)} has markers but not in claimed order (${m} precedes an earlier marker)`,
417
+ }
418
+ }
419
+ last = idx
420
+ }
421
+ return { claim, verified: true, detail: `serial log ${path.basename(logPath)} contains all claimed markers in order` }
422
+ } catch {
423
+ return { claim, verified: false, detail: `could not read serial log ${logPath}` }
424
+ }
425
+ }
426
+
427
+ case "pr_url": {
428
+ // A PR number is only verified by real git/gh evidence: search the
429
+ // last 50 commit subjects for the number. No gh call (avoid a
430
+ // network dependency in a verification path) — git history is the
431
+ // deterministic, offline ground truth for "a PR with this number
432
+ // was actually involved".
433
+ return new Promise((resolve) => {
434
+ execFile(
435
+ "git",
436
+ ["-C", opts.workspaceRoot, "log", "--oneline", "-50"],
437
+ { timeout: 10_000, maxBuffer: 1024 * 1024 },
438
+ (err, stdout) => {
439
+ if (err) {
440
+ resolve({ claim, verified: false, detail: "no git history available to confirm PR evidence" })
441
+ return
442
+ }
443
+ if (stdout.includes(`#${claim.prNumber}`) || stdout.includes(`pull/${claim.prNumber}`)) {
444
+ resolve({ claim, verified: true, detail: `git history references PR #${claim.prNumber}` })
445
+ } else {
446
+ resolve({
447
+ claim,
448
+ verified: false,
449
+ detail: `no git history entry references PR #${claim.prNumber} — no real PR side effect found`,
450
+ })
451
+ }
452
+ },
453
+ )
454
+ })
455
+ }
456
+ }
457
+ }
458
+
459
+ /**
460
+ * Verify a list of extracted claims against ground truth. Returns the full
461
+ * per-claim results; the caller (the loop) treats ANY unverified claim as
462
+ * grounds to defer the completion.
463
+ */
464
+ export async function verifyClaims(
465
+ claims: ExtractedClaim[],
466
+ opts: ClaimVerificationOptions,
467
+ ): Promise<ClaimVerification[]> {
468
+ const results: ClaimVerification[] = []
469
+ for (const claim of claims) {
470
+ // Sequential on purpose: verification re-runs commands that share the
471
+ // host's shell; parallel re-runs could interfere (and the count is
472
+ // small — a completion report has at most a handful of claims).
473
+ results.push(await verifyClaim(claim, opts))
474
+ }
475
+ return results
476
+ }
477
+
478
+ /** Convenience: a stable human-readable label for a claim (for deferral messages). */
479
+ export function claimLabel(claim: ExtractedClaim): string {
480
+ switch (claim.kind) {
481
+ case "file_exists":
482
+ return `the file '${claim.path}' exists`
483
+ case "command_passed":
484
+ return `the command '${claim.command}' passed`
485
+ case "serial_marker":
486
+ return `the serial markers ${claim.markers.join(", ")} appear in order in a serial log`
487
+ case "pr_url":
488
+ return `a real PR #${claim.prNumber} exists`
489
+ }
490
+ }
491
+
492
+ /** Convenience: did a claim set pass verification entirely? */
493
+ export function allClaimsVerified(results: ClaimVerification[]): boolean {
494
+ return results.length > 0 && results.every((r) => r.verified)
495
+ }
496
+
497
+ /** Convenience: the first unverified claim's detail (for the deferral nudge). */
498
+ export function firstUnverifiedDetail(results: ClaimVerification[]): string | undefined {
499
+ return results.find((r) => !r.verified)?.detail
500
+ }
501
+
502
+ /** Re-exported for tests: the workspace-path resolver (kept internal otherwise). */
503
+ export { resolveClaimPath }
@@ -14,6 +14,11 @@
14
14
  * session_start / iteration_start / llm_response / tool_call / tool_result
15
15
  * checkpoint_saved / decision_blocked / decision_answered
16
16
  * condensed / todo_updated / paused / resumed / session_end
17
+ * unverified_claim (fabrication fix, 2026-09-01): emitted when
18
+ * evidenceRequiredCompletion was on and an attempt_completion was DEFERRED
19
+ * because a machine-checkable claim in its result could not be verified
20
+ * against ground truth — carries the verification counts + the first
21
+ * unverified claim's detail (see EventFeed.unverifiedClaim)
17
22
  * attempt_completion (issue #34): the session's FINAL report, emitted when
18
23
  * the loop accepts completion (attempt_completion tool call OR the
19
24
  * text-only success fallback) and carries the FULL report text — this is
@@ -300,6 +305,33 @@ export class EventFeed {
300
305
  })
301
306
  }
302
307
 
308
+ /**
309
+ * Evidence-gated completion (fabrication fix, 2026-09-01): emitted when
310
+ * evidenceRequiredCompletion was on and an attempt_completion was DEFERRED
311
+ * because one or more machine-checkable claims in its result (a file
312
+ * exists, a command passed, serial markers appear, a PR exists) could not
313
+ * be independently verified against ground truth (src/engine/claims.ts).
314
+ * Carries the verification counts + the first unverified claim's detail so
315
+ * downstream consumers (eval, selfplay miner, orchestrator) can see WHY a
316
+ * completion was refused — the structured record that a fabrication
317
+ * attempt was caught, not just a log line.
318
+ */
319
+ unverifiedClaim(fields: {
320
+ iteration: number
321
+ claimsChecked: number
322
+ claimsPassed: number
323
+ claimsUnverified: number
324
+ detail: string
325
+ }): Promise<void> {
326
+ return this.emit("unverified_claim", {
327
+ iteration: fields.iteration,
328
+ claimsChecked: fields.claimsChecked,
329
+ claimsPassed: fields.claimsPassed,
330
+ claimsUnverified: fields.claimsUnverified,
331
+ detail: truncateField(fields.detail).text,
332
+ })
333
+ }
334
+
303
335
  /** iteration 0 = the session's baseline checkpoint (before iteration 1). */
304
336
  checkpointSaved(iteration: number): Promise<void> {
305
337
  return this.emit("checkpoint_saved", { iteration })
@@ -49,7 +49,6 @@ export const CORE_TOOL_NAMES = new Set([
49
49
  "attempt_completion",
50
50
  "execute_command",
51
51
  "list_files",
52
- "new_task",
53
52
  "read_file",
54
53
  // search_replace deliberately excluded — same reasoning as apply_diff
55
54
  // above it in git history: search_replace requires an EXACT literal
@@ -74,9 +73,23 @@ export const CORE_TOOL_NAMES = new Set([
74
73
  // call it correctly but only as narrated text, never having pulled its
75
74
  // real schema in via request_tool first.
76
75
  "set_indentation",
77
- "switch_mode",
78
- "update_todo_list",
79
76
  "write_to_file",
77
+ // 2026-09-02: new_task, switch_mode, and update_todo_list demoted from
78
+ // core to lazy — real usage data across 62 logged local-backend eval
79
+ // sessions (grep "tool result: <name>" across eval_verifier_runs/*/
80
+ // session.log) showed ZERO calls to any of these three, ever, while
81
+ // still paying ~3,700 combined chars of their schemas on every single
82
+ // turn of every session. Unlike set_indentation (which has a specific
83
+ // documented live failure showing the model won't request_tool it when
84
+ // it's actually needed), there is no equivalent evidence for these
85
+ // three — no session in that corpus needed delegation (new_task), a
86
+ // mode switch (switch_mode), or a multi-step plan register
87
+ // (update_todo_list) at all, since these are single-file scratch
88
+ // eval tasks. If a session genuinely needs one, it's still one
89
+ // request_tool call away. Revisit if live data ever shows a session
90
+ // that needed one of these three but never called request_tool for it
91
+ // (the set_indentation failure shape) — that would argue for
92
+ // re-promoting that specific tool back to core.
80
93
  ])
81
94
 
82
95
  export const LIST_TOOLS_NAME = "list_tools"
@@ -111,17 +124,25 @@ export function splitCoreAndLazyTools(allTools: ChatTool[]): SplitTools {
111
124
  }
112
125
 
113
126
  export function buildListToolsTool(): ChatTool {
127
+ // 2026-09-02: built from CORE_TOOL_NAMES itself rather than a hand-
128
+ // written duplicate list — the previous static string had already
129
+ // drifted (it named apply_diff/search_replace as "always available",
130
+ // which was never true; both are deliberately lazy, see
131
+ // CORE_TOOL_NAMES's own comments) and would have drifted again the
132
+ // moment core membership changed without this description changing
133
+ // with it. request_tool/list_tools themselves are the delivery
134
+ // mechanism, not part of the "core" the model chooses among, so they're
135
+ // deliberately left out of this parenthetical (the tool's own name
136
+ // already makes clear it exists).
137
+ const coreList = [...CORE_TOOL_NAMES].join(", ")
114
138
  return {
115
139
  type: "function",
116
140
  function: {
117
141
  name: LIST_TOOLS_NAME,
118
142
  description:
119
- "List additional tools available in this session beyond the core set " +
120
- "(read_file, write_to_file, apply_diff, search_replace, edit_file, " +
121
- "execute_command, list_files, attempt_completion, ask_followup_question, " +
122
- "switch_mode, new_task, update_todo_list — always available, not listed " +
123
- "here). Call request_tool with a name from this list to make that tool " +
124
- "callable on your NEXT turn.",
143
+ `List additional tools available in this session beyond the core set (${coreList} — ` +
144
+ "always available, not listed here). Call request_tool with a name from this list " +
145
+ "to make that tool callable on your NEXT turn.",
125
146
  parameters: { type: "object", properties: {}, required: [] },
126
147
  },
127
148
  }