headlesscode 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -189,6 +189,15 @@ export interface ToolContext {
189
189
  * executors (tests, reviewer/QA).
190
190
  */
191
191
  guardLargeOverwrites?: boolean
192
+ /**
193
+ * See HeadlessSessionConfig.disableReadFileCache (loop.ts) for the full
194
+ * writeup. When true, read_file/list_files always serve real content —
195
+ * the session-scoped "[cache] unchanged, re-read the earlier result"
196
+ * short-circuit (src/tools/executor.ts) never fires. Absent/false for
197
+ * cloud sessions and bare executors (tests, reviewer/QA), where that
198
+ * short-circuit's real token-cost savings still apply.
199
+ */
200
+ disableReadFileCache?: boolean
192
201
  /**
193
202
  * Decision escalation (ask_followup_question, see src/tools/executor.ts):
194
203
  * how long to block waiting for `.harness.decision-answer` before falling
@@ -269,6 +278,24 @@ export interface SessionBudgetUsage {
269
278
  model: string
270
279
  }
271
280
 
281
+ /**
282
+ * Evidence-gated completion (fabrication fix, 2026-09-01): the outcome of
283
+ * the claim-verification pass that ran against the final attempt_completion
284
+ * result when evidenceRequiredCompletion was on (see src/engine/claims.ts).
285
+ * Present on a successful SessionResult ONLY when the gate actually ran and
286
+ * found at least one machine-checkable claim — its presence is the caller's
287
+ * signal that "success" means "claims independently verified against ground
288
+ * truth", not "the model said so". Absent when the gate didn't run.
289
+ */
290
+ export interface SessionCompletionVerification {
291
+ /** Number of machine-checkable claims extracted from the result text. */
292
+ claimsChecked: number
293
+ /** Number of claims independently confirmed against the real world. */
294
+ claimsPassed: number
295
+ /** Number of claims that could not be confirmed (0 on an accepted completion). */
296
+ claimsUnverified: number
297
+ }
298
+
272
299
  export interface SessionResult {
273
300
  status: SessionStatus
274
301
  result?: string
@@ -277,6 +304,15 @@ export interface SessionResult {
277
304
  reason?: string
278
305
  iterations: number
279
306
  toolCalls: number
307
+ /**
308
+ * Evidence-gated completion (fabrication fix): present when the
309
+ * completion's claims were independently verified against ground truth
310
+ * (file existence, real command re-runs, serial logs, git history) — see
311
+ * SessionCompletionVerification. Absent when the gate didn't run (flag
312
+ * off, no claims in the result) — callers must NOT treat that as
313
+ * "verified", only as "not checked".
314
+ */
315
+ verification?: SessionCompletionVerification
280
316
  /**
281
317
  * Issue #34: absolute path to the session's complete final report
282
318
  * (`<workspaceRoot>/.headlesscode/reports/<sessionId>.md`), present when
@@ -0,0 +1,129 @@
1
+ import * as fs from "node:fs/promises"
2
+ import * as path from "node:path"
3
+ import type { CandidateRecord, RsiArchive, RsiRunRecord } from "./types.js"
4
+
5
+ export function archivePath(archiveDir: string): string {
6
+ return path.join(archiveDir, "archive.json")
7
+ }
8
+
9
+ function emptyArchive(): RsiArchive {
10
+ return {
11
+ schemaVersion: 2,
12
+ updatedAt: new Date(0).toISOString(),
13
+ runs: [],
14
+ activeRuns: [],
15
+ candidates: [],
16
+ modelCandidates: [],
17
+ combinations: [],
18
+ jobs: [],
19
+ trajectoryRefs: [],
20
+ curriculumTasks: [],
21
+ }
22
+ }
23
+
24
+ function normalizeRun(run: RsiRunRecord): RsiRunRecord {
25
+ return {
26
+ ...run,
27
+ candidates: Array.isArray(run.candidates) ? run.candidates : [],
28
+ reports: Array.isArray(run.reports) ? run.reports : [],
29
+ selectedCandidates: run.selectedCandidates ?? (run.selected ? [run.selected] : []),
30
+ }
31
+ }
32
+
33
+ function migrate(raw: Partial<RsiArchive> & { schemaVersion?: number }): RsiArchive {
34
+ const archive = emptyArchive()
35
+ archive.updatedAt = raw.updatedAt ?? archive.updatedAt
36
+ archive.runs = (Array.isArray(raw.runs) ? raw.runs : []).map((run) => normalizeRun(run))
37
+ archive.activeRuns = (Array.isArray(raw.activeRuns) ? raw.activeRuns : []).map((run) => normalizeRun(run))
38
+ archive.candidates = Array.isArray(raw.candidates) ? raw.candidates : archive.runs.flatMap((run) => run.candidates)
39
+ archive.modelCandidates = Array.isArray(raw.modelCandidates) ? raw.modelCandidates : []
40
+ archive.combinations = Array.isArray(raw.combinations) ? raw.combinations : []
41
+ archive.jobs = Array.isArray(raw.jobs) ? raw.jobs : []
42
+ archive.trajectoryRefs = Array.isArray(raw.trajectoryRefs) ? raw.trajectoryRefs : []
43
+ archive.curriculumTasks = Array.isArray(raw.curriculumTasks) ? raw.curriculumTasks : []
44
+ return archive
45
+ }
46
+
47
+ export async function readArchive(archiveDir: string): Promise<RsiArchive> {
48
+ try {
49
+ const raw = await fs.readFile(archivePath(archiveDir), "utf8")
50
+ const parsed = JSON.parse(raw) as Record<string, unknown> & { schemaVersion?: number }
51
+ if ((parsed.schemaVersion === 1 || parsed.schemaVersion === 2) && Array.isArray(parsed.runs)) {
52
+ return migrate(parsed as Partial<RsiArchive> & { schemaVersion?: number })
53
+ }
54
+ } catch {
55
+ // A missing or malformed archive starts a new durable history. The next
56
+ // write makes the state explicit and inspectable.
57
+ }
58
+ return emptyArchive()
59
+ }
60
+
61
+ export async function writeArchive(archiveDir: string, archive: RsiArchive): Promise<void> {
62
+ await fs.mkdir(archiveDir, { recursive: true })
63
+ const destination = archivePath(archiveDir)
64
+ const temporary = `${destination}.tmp-${process.pid}`
65
+ await fs.writeFile(temporary, `${JSON.stringify(archive, null, 2)}\n`, "utf8")
66
+ await fs.rename(temporary, destination)
67
+ }
68
+
69
+ export async function appendRun(archiveDir: string, run: RsiRunRecord): Promise<RsiArchive> {
70
+ const archive = await readArchive(archiveDir)
71
+ archive.activeRuns = archive.activeRuns.filter((entry) => entry.runId !== run.runId)
72
+ archive.runs = [...archive.runs.filter((entry) => entry.runId !== run.runId), normalizeRun(run)]
73
+ for (const candidate of run.candidates) {
74
+ const index = archive.candidates.findIndex((entry) => entry.id === candidate.id)
75
+ if (index >= 0) archive.candidates[index] = candidate
76
+ else archive.candidates.push(candidate)
77
+ }
78
+ for (const model of run.modelCandidates ?? []) {
79
+ if (!archive.modelCandidates.some((entry) => entry.id === model.id)) archive.modelCandidates.push(model)
80
+ }
81
+ for (const combination of run.combinations ?? []) {
82
+ const index = archive.combinations.findIndex((entry) => entry.id === combination.id)
83
+ if (index >= 0) archive.combinations[index] = combination
84
+ else archive.combinations.push(combination)
85
+ }
86
+ for (const job of run.jobs ?? []) {
87
+ const index = archive.jobs.findIndex((entry) => entry.id === job.id)
88
+ if (index >= 0) archive.jobs[index] = job
89
+ else archive.jobs.push(job)
90
+ }
91
+ for (const task of run.curriculumTasks ?? []) {
92
+ if (!archive.curriculumTasks.some((entry) => entry.id === task.id)) archive.curriculumTasks.push(task)
93
+ }
94
+ archive.trajectoryRefs = [...new Set([...archive.trajectoryRefs, ...(run.trajectoryRefs ?? [])])]
95
+ archive.updatedAt = new Date().toISOString()
96
+ await writeArchive(archiveDir, archive)
97
+ return archive
98
+ }
99
+
100
+ export async function checkpointRun(archiveDir: string, run: RsiRunRecord): Promise<RsiArchive> {
101
+ const archive = await readArchive(archiveDir)
102
+ const normalized = normalizeRun(run)
103
+ archive.activeRuns = [...archive.activeRuns.filter((entry) => entry.runId !== run.runId), normalized]
104
+ for (const candidate of run.candidates) {
105
+ const index = archive.candidates.findIndex((entry) => entry.id === candidate.id)
106
+ if (index >= 0) archive.candidates[index] = candidate
107
+ else archive.candidates.push(candidate)
108
+ }
109
+ for (const job of run.jobs ?? []) {
110
+ const index = archive.jobs.findIndex((entry) => entry.id === job.id)
111
+ if (index >= 0) archive.jobs[index] = job
112
+ else archive.jobs.push(job)
113
+ }
114
+ archive.updatedAt = new Date().toISOString()
115
+ await writeArchive(archiveDir, archive)
116
+ return archive
117
+ }
118
+
119
+ export async function findActiveRun(archiveDir: string, runId: string): Promise<RsiRunRecord | undefined> {
120
+ return (await readArchive(archiveDir)).activeRuns.find((run) => run.runId === runId)
121
+ }
122
+
123
+ export function archiveCandidate(archive: RsiArchive, candidate: CandidateRecord): RsiArchive {
124
+ const index = archive.candidates.findIndex((entry) => entry.id === candidate.id)
125
+ if (index >= 0) archive.candidates[index] = candidate
126
+ else archive.candidates.push(candidate)
127
+ archive.updatedAt = new Date().toISOString()
128
+ return archive
129
+ }
@@ -0,0 +1,312 @@
1
+ import * as path from "node:path"
2
+ import { DEFAULT_PROTECTED_FILES } from "../permissions/protected-files.js"
3
+ import type { ComputePolicy, MutationHypothesis, MutationKind, ParentSelectionPolicy, RsiConfig } from "./types.js"
4
+
5
+ export const DEFAULT_RSI_MODEL = "wxrq-qwen3.5-9b:latest"
6
+ export const DEFAULT_VISIBLE_EVALS = ["npm test -- --filter rsi"]
7
+ export const DEFAULT_PROTECTED_PATHS = [
8
+ ...DEFAULT_PROTECTED_FILES,
9
+ "src/rsi/",
10
+ "scripts/",
11
+ "**/*.test.ts",
12
+ "package.json",
13
+ "package-lock.json",
14
+ ".gitignore",
15
+ "scripts/eval-suite/",
16
+ ".headlesscode/",
17
+ ".worktrees/",
18
+ ]
19
+
20
+ export interface ParsedRsiArgs {
21
+ config?: RsiConfig
22
+ help?: boolean
23
+ error?: string
24
+ }
25
+
26
+ function positiveInteger(value: string, name: string): number {
27
+ const parsed = Number(value)
28
+ if (!Number.isInteger(parsed) || parsed < 1) {
29
+ throw new Error(`${name} must be a positive integer`)
30
+ }
31
+ return parsed
32
+ }
33
+
34
+ function splitCommands(value: string): string[] {
35
+ return value
36
+ .split(";;")
37
+ .map((command) => command.trim())
38
+ .filter(Boolean)
39
+ }
40
+
41
+ export function rsiHelp(): string {
42
+ return `headlesscode improve — bounded recursive self-improvement
43
+
44
+ Usage:
45
+ headlesscode improve --repo <path> [options]
46
+
47
+ Options:
48
+ --model <id> worker model (default: ${DEFAULT_RSI_MODEL})
49
+ --population <n> candidates per generation (default: 2)
50
+ --generations <n> bounded generations (default: 1)
51
+ --parent-policy <name> champion-specialist-novelty | pareto-front | all-eligible
52
+ --mutation-kind <name> corrective | architectural | search-policy | curriculum | model-adaptation
53
+ --hypothesis <text> recorded reason for the mutation
54
+ --expected-effect <text> measurable benefit to test
55
+ --potential-downside <text> recorded cost or regression risk
56
+ --compute-policy <name> single | independent | planner-executors | critic-retry
57
+ --eval <command> visible command; repeat or separate with ;;
58
+ --hidden-eval <command> supervisor-only hidden command
59
+ --mutation-task <text> improvement objective
60
+ --archive-dir <path> durable archive directory
61
+ --worktree-dir <path> candidate worktree directory
62
+ --trajectory-dir <path> structured trajectory output directory
63
+ --curriculum-dir <path> generated curriculum proposal directory
64
+ --resume <run-id> resume a checkpointed active run
65
+ --model-candidate-id <id> model identity used for combination tracking
66
+ --base-ref <ref> git ref to branch from (default: HEAD)
67
+ --dry-run print the planned loop without changing files
68
+ --keep-worktrees retain candidate worktrees after evaluation
69
+ --help show this help
70
+
71
+ The evaluator and scoring code are protected from candidates. The first run
72
+ uses the local Qwen worker unless --model overrides it.
73
+ `
74
+ }
75
+
76
+ export function parseRsiArgs(argv: string[], cwd = process.cwd()): ParsedRsiArgs {
77
+ let repoRoot = cwd
78
+ let model = DEFAULT_RSI_MODEL
79
+ let population = 2
80
+ let generations = 1
81
+ let maxConcurrent = 1
82
+ let mutationTask = "Improve the headlesscode agent loop while preserving all existing behavior and tests."
83
+ let parentSelectionPolicy: ParentSelectionPolicy = "champion-specialist-novelty"
84
+ let mutationKind: MutationKind = "corrective"
85
+ let hypothesis: MutationHypothesis = {
86
+ statement: mutationTask,
87
+ expectedEffect: "higher verified task success without regressions",
88
+ potentialDownside: "additional complexity or inference cost",
89
+ }
90
+ let computePolicy: ComputePolicy = "single"
91
+ let evalCommands = [...DEFAULT_VISIBLE_EVALS]
92
+ let hiddenEvalCommands: string[] = []
93
+ let archiveDir: string | undefined
94
+ let worktreeDir: string | undefined
95
+ let baseRef = "HEAD"
96
+ let seed = "headlesscode-rsi"
97
+ let dryRun = false
98
+ let keepWorktrees = false
99
+ let maxIterations = 40
100
+ let commandTimeoutMs = 15 * 60_000
101
+ let trajectoryDir: string | undefined
102
+ let curriculumDir: string | undefined
103
+ let resumeRunId: string | undefined
104
+ let modelCandidateId: string | undefined
105
+
106
+ const take = (index: number, name: string): [string, number] => {
107
+ const value = argv[index + 1]
108
+ if (!value || value.startsWith("--")) {
109
+ throw new Error(`${name} requires a value`)
110
+ }
111
+ return [value, index + 1]
112
+ }
113
+
114
+ try {
115
+ for (let index = 0; index < argv.length; index++) {
116
+ const arg = argv[index]
117
+ switch (arg) {
118
+ case "--help":
119
+ case "-h":
120
+ return { help: true }
121
+ case "--repo": {
122
+ const [value, next] = take(index, arg)
123
+ repoRoot = path.resolve(value)
124
+ index = next
125
+ break
126
+ }
127
+ case "--model": {
128
+ const [value, next] = take(index, arg)
129
+ model = value
130
+ index = next
131
+ break
132
+ }
133
+ case "--population": {
134
+ const [value, next] = take(index, arg)
135
+ population = positiveInteger(value, arg)
136
+ index = next
137
+ break
138
+ }
139
+ case "--generations": {
140
+ const [value, next] = take(index, arg)
141
+ generations = positiveInteger(value, arg)
142
+ index = next
143
+ break
144
+ }
145
+ case "--parent-policy": {
146
+ const [value, next] = take(index, arg)
147
+ if (!["champion-specialist-novelty", "pareto-front", "all-eligible"].includes(value)) throw new Error(`${arg} has an invalid policy`)
148
+ parentSelectionPolicy = value as ParentSelectionPolicy
149
+ index = next
150
+ break
151
+ }
152
+ case "--mutation-kind": {
153
+ const [value, next] = take(index, arg)
154
+ if (!["corrective", "architectural", "search-policy", "curriculum", "model-adaptation"].includes(value)) throw new Error(`${arg} has an invalid mutation kind`)
155
+ mutationKind = value as MutationKind
156
+ hypothesis.statement = mutationTask
157
+ index = next
158
+ break
159
+ }
160
+ case "--hypothesis": {
161
+ const [value, next] = take(index, arg)
162
+ hypothesis.statement = value
163
+ index = next
164
+ break
165
+ }
166
+ case "--expected-effect": {
167
+ const [value, next] = take(index, arg)
168
+ hypothesis.expectedEffect = value
169
+ index = next
170
+ break
171
+ }
172
+ case "--potential-downside": {
173
+ const [value, next] = take(index, arg)
174
+ hypothesis.potentialDownside = value
175
+ index = next
176
+ break
177
+ }
178
+ case "--compute-policy": {
179
+ const [value, next] = take(index, arg)
180
+ if (!["single", "independent", "planner-executors", "critic-retry"].includes(value)) throw new Error(`${arg} has an invalid policy`)
181
+ computePolicy = value as ComputePolicy
182
+ index = next
183
+ break
184
+ }
185
+ case "--max-concurrent": {
186
+ const [value, next] = take(index, arg)
187
+ maxConcurrent = positiveInteger(value, arg)
188
+ index = next
189
+ break
190
+ }
191
+ case "--eval": {
192
+ const [value, next] = take(index, arg)
193
+ evalCommands = splitCommands(value)
194
+ index = next
195
+ break
196
+ }
197
+ case "--hidden-eval": {
198
+ const [value, next] = take(index, arg)
199
+ hiddenEvalCommands.push(...splitCommands(value))
200
+ index = next
201
+ break
202
+ }
203
+ case "--mutation-task": {
204
+ const [value, next] = take(index, arg)
205
+ mutationTask = value
206
+ index = next
207
+ break
208
+ }
209
+ case "--archive-dir": {
210
+ const [value, next] = take(index, arg)
211
+ archiveDir = path.resolve(repoRoot, value)
212
+ index = next
213
+ break
214
+ }
215
+ case "--worktree-dir": {
216
+ const [value, next] = take(index, arg)
217
+ worktreeDir = path.resolve(repoRoot, value)
218
+ index = next
219
+ break
220
+ }
221
+ case "--trajectory-dir": {
222
+ const [value, next] = take(index, arg)
223
+ trajectoryDir = path.resolve(repoRoot, value)
224
+ index = next
225
+ break
226
+ }
227
+ case "--curriculum-dir": {
228
+ const [value, next] = take(index, arg)
229
+ curriculumDir = path.resolve(repoRoot, value)
230
+ index = next
231
+ break
232
+ }
233
+ case "--resume": {
234
+ const [value, next] = take(index, arg)
235
+ resumeRunId = value
236
+ index = next
237
+ break
238
+ }
239
+ case "--model-candidate-id": {
240
+ const [value, next] = take(index, arg)
241
+ modelCandidateId = value
242
+ index = next
243
+ break
244
+ }
245
+ case "--base-ref": {
246
+ const [value, next] = take(index, arg)
247
+ baseRef = value
248
+ index = next
249
+ break
250
+ }
251
+ case "--seed": {
252
+ const [value, next] = take(index, arg)
253
+ seed = value
254
+ index = next
255
+ break
256
+ }
257
+ case "--max-iterations": {
258
+ const [value, next] = take(index, arg)
259
+ maxIterations = positiveInteger(value, arg)
260
+ index = next
261
+ break
262
+ }
263
+ case "--timeout-ms": {
264
+ const [value, next] = take(index, arg)
265
+ commandTimeoutMs = positiveInteger(value, arg)
266
+ index = next
267
+ break
268
+ }
269
+ case "--dry-run":
270
+ dryRun = true
271
+ break
272
+ case "--keep-worktrees":
273
+ keepWorktrees = true
274
+ break
275
+ default:
276
+ throw new Error(`unknown improve argument: ${arg}`)
277
+ }
278
+ }
279
+ const resolvedRepo = path.resolve(repoRoot)
280
+ return {
281
+ config: {
282
+ repoRoot: resolvedRepo,
283
+ model,
284
+ population,
285
+ generations,
286
+ maxConcurrent,
287
+ mutationTask,
288
+ parentSelectionPolicy,
289
+ mutationKind,
290
+ hypothesis: { ...hypothesis },
291
+ computePolicy,
292
+ evalCommands,
293
+ hiddenEvalCommands,
294
+ archiveDir: archiveDir ?? path.join(resolvedRepo, ".headlesscode", "rsi"),
295
+ worktreeDir: worktreeDir ?? path.join(resolvedRepo, ".worktrees", "rsi"),
296
+ trajectoryDir: trajectoryDir ?? path.join(archiveDir ?? path.join(resolvedRepo, ".headlesscode", "rsi"), "trajectories"),
297
+ curriculumDir: curriculumDir ?? path.join(archiveDir ?? path.join(resolvedRepo, ".headlesscode", "rsi"), "curriculum"),
298
+ baseRef,
299
+ seed,
300
+ dryRun,
301
+ keepWorktrees,
302
+ maxIterations,
303
+ protectedPaths: [...DEFAULT_PROTECTED_PATHS],
304
+ commandTimeoutMs,
305
+ resumeRunId,
306
+ modelCandidateId,
307
+ },
308
+ }
309
+ } catch (error) {
310
+ return { error: error instanceof Error ? error.message : String(error) }
311
+ }
312
+ }