headlesscode 1.0.3 → 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,317 @@
1
+ import type { ChildProcess } from "node:child_process"
2
+
3
+ export type CandidateStatus = "planned" | "mutating" | "evaluating" | "accepted" | "rejected" | "failed"
4
+ export type MutationKind = "corrective" | "architectural" | "search-policy" | "curriculum" | "model-adaptation"
5
+ export type ParentSelectionPolicy = "champion-specialist-novelty" | "pareto-front" | "all-eligible"
6
+ export type ComputePolicy = "single" | "independent" | "planner-executors" | "critic-retry"
7
+
8
+ export interface MutationHypothesis {
9
+ statement: string
10
+ expectedEffect: string
11
+ potentialDownside: string
12
+ evidence?: string
13
+ }
14
+
15
+ export interface MetricVector {
16
+ correctness: number
17
+ reliability: number
18
+ generalization: number
19
+ hidden: number
20
+ efficiency: number
21
+ latency: number
22
+ tokenUse: number
23
+ recovery: number
24
+ fabricationRate: number
25
+ complexityPenalty: number
26
+ }
27
+
28
+ export interface ComplexityMetrics {
29
+ diffLines: number
30
+ changedFiles: number
31
+ newDependencies: number
32
+ additionalModelCalls: number
33
+ runtimeOverheadMs: number
34
+ }
35
+
36
+ export interface ParentSelectionReason {
37
+ strategy: "champion" | "specialist" | "novelty" | "pareto" | "archive" | "baseline"
38
+ reason: string
39
+ metrics?: Partial<MetricVector>
40
+ }
41
+
42
+ export type ModelRole =
43
+ | "worker"
44
+ | "mutation-architect"
45
+ | "failure-analyst"
46
+ | "critic"
47
+ | "adversary"
48
+ | "reviewer"
49
+ | "curriculum-designer"
50
+ | "training-data-curator"
51
+
52
+ export interface RoleModelConfig {
53
+ provider: "ollama" | "openrouter" | "command"
54
+ model: string
55
+ baseUrl?: string
56
+ command?: string
57
+ }
58
+
59
+ export type RsiRoleConfig = Partial<Record<ModelRole, RoleModelConfig>>
60
+
61
+ export interface TrainingConfig {
62
+ method: "none" | "lora" | "qlora" | "sft" | "preference"
63
+ command?: string
64
+ seed?: number
65
+ parameters?: Record<string, string | number | boolean>
66
+ }
67
+
68
+ export interface ModelCandidate {
69
+ id: string
70
+ parentModel: string
71
+ trainingMethod: TrainingConfig["method"]
72
+ datasetVersion?: string
73
+ trainingConfig: TrainingConfig
74
+ artifactPath?: string
75
+ artifactHash?: string
76
+ status: "base" | "prepared" | "trained" | "evaluated" | "accepted" | "rejected"
77
+ createdAt: string
78
+ provenance?: Record<string, string>
79
+ }
80
+
81
+ export interface HarnessModelCombination {
82
+ id: string
83
+ harnessCandidateId: string
84
+ modelCandidateId: string
85
+ status: "planned" | "evaluated" | "accepted" | "rejected"
86
+ metrics?: MetricVector
87
+ fitness?: Fitness
88
+ createdAt: string
89
+ }
90
+
91
+ export type ResourceClass = "LOCAL_GPU" | "CPU" | "REMOTE_API" | "TRAINING_GPU"
92
+ export type ExperimentJobKind = "mutation" | "evaluation" | "adversarial" | "curriculum" | "training" | "model-evaluation" | "critic"
93
+ export type ExperimentJobStatus = "queued" | "running" | "completed" | "failed" | "cancelled"
94
+
95
+ export interface ResourceRequirements {
96
+ class: ResourceClass
97
+ units: number
98
+ concurrencyKey?: string
99
+ }
100
+
101
+ export interface ExperimentJob {
102
+ id: string
103
+ kind: ExperimentJobKind
104
+ status: ExperimentJobStatus
105
+ resource: ResourceRequirements
106
+ owner?: string
107
+ candidateId?: string
108
+ createdAt: string
109
+ updatedAt: string
110
+ attempts: number
111
+ error?: string
112
+ }
113
+
114
+ export interface TrajectoryMessage {
115
+ role: string
116
+ content?: string
117
+ tool_calls?: unknown[]
118
+ tool_call_id?: string
119
+ name?: string
120
+ }
121
+
122
+ export interface TrajectoryRecord {
123
+ id: string
124
+ task: string
125
+ environment: { repoRoot: string; baseCommit: string; generation: number }
126
+ model: { id: string; provider?: string; candidateId?: string }
127
+ harnessVersion: string
128
+ promptConfig: Record<string, unknown>
129
+ messages: TrajectoryMessage[]
130
+ toolCalls: number
131
+ outcome: "success" | "failure" | "incomplete"
132
+ verification: { verified: boolean; regressionPass: boolean; hiddenPass: boolean }
133
+ failureClassification?: string
134
+ criticDiagnosis?: string
135
+ fitnessImpact?: number
136
+ provenance: { source: string; capturedAt: string; trusted: boolean }
137
+ }
138
+
139
+ export interface TrajectorySummary {
140
+ path: string
141
+ messageCount: number
142
+ toolCalls: number
143
+ trusted: boolean
144
+ outcome: TrajectoryRecord["outcome"]
145
+ failureClassification?: string
146
+ }
147
+
148
+ export interface CurriculumTask {
149
+ id: string
150
+ task: string
151
+ difficulty: 1 | 2 | 3 | 4 | 5
152
+ capability: string
153
+ groundTruthCommand: string
154
+ provenance: { sourceCandidateIds: string[]; failureClass: string; generatedAt: string }
155
+ validated: boolean
156
+ }
157
+
158
+ export interface RsiConfig {
159
+ repoRoot: string
160
+ model: string
161
+ population: number
162
+ generations: number
163
+ maxConcurrent: number
164
+ mutationTask: string
165
+ evalCommands: string[]
166
+ hiddenEvalCommands: string[]
167
+ archiveDir: string
168
+ worktreeDir: string
169
+ baseRef: string
170
+ seed: string
171
+ dryRun: boolean
172
+ keepWorktrees: boolean
173
+ maxIterations: number
174
+ protectedPaths: string[]
175
+ commandTimeoutMs: number
176
+ parentSelectionPolicy?: ParentSelectionPolicy
177
+ eliteCount?: number
178
+ specialistCount?: number
179
+ noveltyRate?: number
180
+ mutationKind?: MutationKind
181
+ hypothesis?: MutationHypothesis
182
+ computePolicy?: ComputePolicy
183
+ modelCandidateId?: string
184
+ modelCandidates?: ModelCandidate[]
185
+ roles?: RsiRoleConfig
186
+ trajectoryDir?: string
187
+ curriculumDir?: string
188
+ resumeRunId?: string
189
+ }
190
+
191
+ export interface CandidateRecord {
192
+ id: string
193
+ generation: number
194
+ parent: string
195
+ branch: string
196
+ worktree: string
197
+ baseCommit: string
198
+ status: CandidateStatus
199
+ model: string
200
+ mutation: string
201
+ createdAt: string
202
+ updatedAt: string
203
+ commits: string[]
204
+ changedFiles: string[]
205
+ protectedPathViolations: string[]
206
+ mutationKind?: MutationKind
207
+ hypothesis?: MutationHypothesis
208
+ parentSelection?: ParentSelectionReason
209
+ parentCommit?: string
210
+ modelCandidateId?: string
211
+ combinationId?: string
212
+ failure?: string
213
+ result?: TrialResult
214
+ fitness?: Fitness
215
+ trajectory?: TrajectorySummary
216
+ }
217
+
218
+ export interface TrialResult {
219
+ ok: boolean
220
+ command: string
221
+ exitCode: number | null
222
+ durationMs: number
223
+ stdout: string
224
+ stderr: string
225
+ timedOut?: boolean
226
+ }
227
+
228
+ export interface EvaluationSummary {
229
+ regression: TrialResult
230
+ visible: TrialResult[]
231
+ hidden: TrialResult[]
232
+ completed: boolean
233
+ crashed: boolean
234
+ protectedPathViolation: boolean
235
+ changedFiles: string[]
236
+ committed?: boolean
237
+ complexity?: ComplexityMetrics
238
+ failureClassification?: string
239
+ }
240
+
241
+ export interface HardGates {
242
+ regressionPass: boolean
243
+ visibleEvalPass: boolean
244
+ hiddenEvalPass: boolean
245
+ noProtectedPathViolation: boolean
246
+ completed: boolean
247
+ noCrash: boolean
248
+ committed: boolean
249
+ }
250
+
251
+ export interface Fitness {
252
+ score: number
253
+ components: {
254
+ regression: number
255
+ visible: number
256
+ hidden: number
257
+ efficiency: number
258
+ recovery: number
259
+ }
260
+ metrics: MetricVector
261
+ complexity: ComplexityMetrics
262
+ paretoRank?: number
263
+ hardGates: HardGates
264
+ reason: string
265
+ }
266
+
267
+ export interface RsiRunRecord {
268
+ runId: string
269
+ startedAt: string
270
+ finishedAt?: string
271
+ baseline?: TrialResult
272
+ model: string
273
+ baseRef: string
274
+ baseCommit: string
275
+ generations: number
276
+ selected?: string
277
+ selectedCandidates?: string[]
278
+ parentSelectionPolicy?: ParentSelectionPolicy
279
+ parentChoices?: Record<string, ParentSelectionReason>
280
+ modelCandidates?: ModelCandidate[]
281
+ combinations?: HarnessModelCombination[]
282
+ jobs?: ExperimentJob[]
283
+ trajectoryRefs?: string[]
284
+ curriculumTasks?: CurriculumTask[]
285
+ candidates: CandidateRecord[]
286
+ reports: string[]
287
+ }
288
+
289
+ export interface RsiArchive {
290
+ schemaVersion: 2
291
+ updatedAt: string
292
+ runs: RsiRunRecord[]
293
+ activeRuns: RsiRunRecord[]
294
+ candidates: CandidateRecord[]
295
+ modelCandidates: ModelCandidate[]
296
+ combinations: HarnessModelCombination[]
297
+ jobs: ExperimentJob[]
298
+ trajectoryRefs: string[]
299
+ curriculumTasks: CurriculumTask[]
300
+ }
301
+
302
+ export interface CommandRunner {
303
+ (command: string, cwd: string, timeoutMs: number, env?: NodeJS.ProcessEnv): Promise<TrialResult>
304
+ }
305
+
306
+ export interface MutationRunner {
307
+ (candidate: CandidateRecord, config: RsiConfig): Promise<{ ok: boolean; result?: TrialResult; error?: string }>
308
+ }
309
+
310
+ export interface RsiHooks {
311
+ runCommand?: CommandRunner
312
+ runMutation?: MutationRunner
313
+ now?: () => string
314
+ log?: (line: string) => void
315
+ }
316
+
317
+ export type SpawnedProcess = ChildProcess
@@ -0,0 +1,96 @@
1
+ import { execFile as execFileCallback } from "node:child_process"
2
+ import * as fs from "node:fs/promises"
3
+ import * as path from "node:path"
4
+ import { promisify } from "node:util"
5
+ import { modelCombinationId } from "./models.js"
6
+ import type { CandidateRecord, RsiConfig } from "./types.js"
7
+ import type { ParentChoice } from "./selection.js"
8
+
9
+ const execFile = promisify(execFileCallback)
10
+
11
+ export function candidateId(generation: number, index: number, seed: string): string {
12
+ const cleanSeed = seed.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "") || "run"
13
+ return `g${generation}-c${String(index + 1).padStart(2, "0")}-${cleanSeed.slice(0, 20)}`
14
+ }
15
+
16
+ export function candidateBranch(id: string): string {
17
+ return `rsi/${id}`
18
+ }
19
+
20
+ export function candidateWorktree(config: RsiConfig, id: string): string {
21
+ return path.join(config.worktreeDir, id)
22
+ }
23
+
24
+ export async function gitOutput(repoRoot: string, args: string[]): Promise<string> {
25
+ const result = await execFile("git", args, { cwd: repoRoot, maxBuffer: 4 * 1024 * 1024 })
26
+ return result.stdout.trim()
27
+ }
28
+
29
+ export async function resolveBaseCommit(config: RsiConfig): Promise<string> {
30
+ return gitOutput(config.repoRoot, ["rev-parse", `${config.baseRef}^{commit}`])
31
+ }
32
+
33
+ export function newCandidate(
34
+ config: RsiConfig,
35
+ generation: number,
36
+ index: number,
37
+ baseCommit: string,
38
+ now: string,
39
+ parentChoice?: ParentChoice,
40
+ ): CandidateRecord {
41
+ const id = candidateId(generation, index, config.seed)
42
+ const modelCandidateId = config.modelCandidateId ?? "model-base"
43
+ return {
44
+ id,
45
+ generation,
46
+ parent: parentChoice?.candidateId ?? (generation === 0 ? "baseline" : "champion"),
47
+ branch: candidateBranch(id),
48
+ worktree: candidateWorktree(config, id),
49
+ baseCommit,
50
+ status: "planned",
51
+ model: config.model,
52
+ mutation: config.mutationTask,
53
+ createdAt: now,
54
+ updatedAt: now,
55
+ commits: [],
56
+ changedFiles: [],
57
+ protectedPathViolations: [],
58
+ mutationKind: config.mutationKind ?? "corrective",
59
+ hypothesis: config.hypothesis,
60
+ parentSelection: parentChoice?.reason ?? (generation === 0 ? { strategy: "baseline", reason: "initial base ref" } : undefined),
61
+ parentCommit: parentChoice?.baseCommit ?? baseCommit,
62
+ modelCandidateId,
63
+ combinationId: modelCombinationId(id, modelCandidateId),
64
+ }
65
+ }
66
+
67
+ export async function createCandidateWorktree(config: RsiConfig, candidate: CandidateRecord): Promise<void> {
68
+ await fs.mkdir(config.worktreeDir, { recursive: true })
69
+ await execFile("git", ["worktree", "add", "-b", candidate.branch, candidate.worktree, candidate.baseCommit], {
70
+ cwd: config.repoRoot,
71
+ maxBuffer: 4 * 1024 * 1024,
72
+ })
73
+ }
74
+
75
+ export async function removeCandidateWorktree(config: RsiConfig, candidate: CandidateRecord): Promise<void> {
76
+ await execFile("git", ["worktree", "remove", "--force", candidate.worktree], {
77
+ cwd: config.repoRoot,
78
+ maxBuffer: 4 * 1024 * 1024,
79
+ })
80
+ }
81
+
82
+ export async function changedFiles(repoRoot: string, baseCommit: string): Promise<string[]> {
83
+ const output = await gitOutput(repoRoot, ["diff", "--name-only", `${baseCommit}...HEAD`])
84
+ const status = await gitOutput(repoRoot, ["status", "--short"])
85
+ const files = new Set(output.split("\n").map((file) => file.trim()).filter(Boolean))
86
+ for (const line of status.split("\n")) {
87
+ const file = line.slice(3).trim()
88
+ if (file) files.add(file)
89
+ }
90
+ return [...files].sort()
91
+ }
92
+
93
+ export async function candidateCommits(repoRoot: string, baseCommit: string): Promise<string[]> {
94
+ const output = await gitOutput(repoRoot, ["log", "--format=%H", `${baseCommit}..HEAD`])
95
+ return output.split("\n").map((commit) => commit.trim()).filter(Boolean)
96
+ }
@@ -190,6 +190,8 @@ export interface ToolExecutorOptions {
190
190
  permissions?: PermissionsConfig
191
191
  /** See ToolContext.guardLargeOverwrites (types.ts) for the full writeup. */
192
192
  guardLargeOverwrites?: boolean
193
+ /** See ToolContext.disableReadFileCache (types.ts) for the full writeup. */
194
+ disableReadFileCache?: boolean
193
195
  /**
194
196
  * Live worker monitoring: fired at the same lifecycle points where the
195
197
  * `.harness.needs-decision` marker is written/cleared, so the session can
@@ -359,6 +361,7 @@ export class ToolExecutor {
359
361
  workspaceRoot: this.workspaceRoot,
360
362
  permissions: this.permissions,
361
363
  guardLargeOverwrites: this.options.guardLargeOverwrites,
364
+ disableReadFileCache: this.options.disableReadFileCache,
362
365
  decisionTimeoutMs: this.options.decisionTimeoutMs,
363
366
  decisionPollIntervalMs: this.options.decisionPollIntervalMs,
364
367
  pauseBudgetClock: this.pauseBudgetClock,
@@ -799,7 +802,32 @@ function readFileHandler(
799
802
  // cached hash can be reused without re-reading the file; any mismatch
800
803
  // (including a same-length rewrite, which changes mtime) falls back to
801
804
  // a full read + sha256 below.
805
+ //
806
+ // 2026-09-02: real, confirmed, live-observed failure mode of this
807
+ // short-circuit against a local model -- ctx.disableReadFileCache
808
+ // skips both cache-hit checks in this function entirely, always
809
+ // serving real content. A session's edit_file call failed ("no
810
+ // match found"), the error told it to re-read and retry, it DID
811
+ // call read_file again exactly as instructed, and got back
812
+ // "[cache] this file is unchanged... re-read the earlier tool
813
+ // result" instead of the actual content -- correct per this
814
+ // mechanism's own design (the safety valve is "the SECOND
815
+ // consecutive identical call serves real content again"), but the
816
+ // model never made that second call: it read the cache-hit message
817
+ // as "you already have what you need", gave up, and called
818
+ // attempt_completion claiming the endpoint worked -- a genuine
819
+ // fabrication directly caused by this response, not a model
820
+ // hallucination from nothing. This short-circuit's own rationale
821
+ // (avoid paying full token cost for content the conversation
822
+ // already has) is a real concern for a REMOTE model's per-token API
823
+ // bill; re-serving a few hundred lines of file content costs a
824
+ // local session near-nothing (it's prefill, not generation, so it
825
+ // barely affects wall-clock time either) against a GPU with no
826
+ // per-token price. Cheap insurance against a much more expensive
827
+ // failure mode locally; the cloud sessions this genuinely saves
828
+ // money for are unaffected (disableReadFileCache stays unset there).
802
829
  if (
830
+ !ctx.disableReadFileCache &&
803
831
  cache !== undefined &&
804
832
  !cache.toldUnchanged &&
805
833
  cache.size === stat.size &&
@@ -818,8 +846,9 @@ function readFileHandler(
818
846
 
819
847
  // Cache-check the CURRENT on-disk content (never "no write tool was
820
848
  // called"): identical args + identical hash => byte-identical output.
849
+ // See the disableReadFileCache comment on the check above.
821
850
  const currentHash = hashFileContent(content)
822
- if (cache !== undefined && cache.hash === currentHash && !cache.toldUnchanged) {
851
+ if (!ctx.disableReadFileCache && cache !== undefined && cache.hash === currentHash && !cache.toldUnchanged) {
823
852
  cache.toldUnchanged = true
824
853
  return ok(READ_FILE_CACHE_HIT_MESSAGE)
825
854
  }
@@ -1565,9 +1594,48 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
1565
1594
 
1566
1595
  return (async () => {
1567
1596
  let cwd = ctx.workspaceRoot
1568
- if (args.cwd != null && args.cwd !== "") {
1597
+ // 2026-09-02: real, confirmed, fully deterministic bug -- a local
1598
+ // LoRA backend's tool-call parser was found sending the literal
1599
+ // 4-character STRING "null" (or "None") for an unset optional
1600
+ // `cwd`, not a real absent/null value. `args.cwd != null` is only
1601
+ // false for the JS primitive null -- a non-empty string "null"
1602
+ // sails through this check as if it were a real requested cwd,
1603
+ // resolving to "<workspaceRoot>/null", which (almost) never
1604
+ // exists. That was fixed at the source (the LoRA server's own
1605
+ // parser), but defend here too: any backend/model can make the
1606
+ // same JSON-serialization slip, and treating the literal text
1607
+ // null/None the same as a real null is cheap and unambiguous --
1608
+ // no real directory is ever named exactly "null" or "None".
1609
+ const isNullish = args.cwd == null || args.cwd === "" || args.cwd === "null" || args.cwd === "None"
1610
+ if (!isNullish) {
1569
1611
  const cwdArg = requireString(args, "cwd")
1570
1612
  cwd = safeTarget(ctx, cwdArg)
1613
+ // Validate BEFORE spawning rather than let a bad cwd reach
1614
+ // spawn(): Node's child_process misattributes a chdir()
1615
+ // failure (nonexistent cwd) to the SHELL itself -- "spawn
1616
+ // /bin/bash ENOENT" -- with no mention of cwd anywhere in the
1617
+ // error, which is exactly the "infrastructure spawn failure"
1618
+ // pattern several ground-truth eval runs hit today (verified
1619
+ // directly: spawning with a nonexistent cwd reproduces that
1620
+ // precise error string/code/path). That misattribution isn't
1621
+ // specific to the null-string bug above -- ANY nonexistent
1622
+ // cwd the model supplies (a stale/hallucinated path) hits it
1623
+ // the same way. Checking here turns a misleading spawn crash
1624
+ // into a clear, actionable, model-facing error instead, and
1625
+ // correctly counts it as a real mistake (it's the model's
1626
+ // bad path, not infrastructure).
1627
+ let cwdStat: fs.Stats | undefined
1628
+ try {
1629
+ cwdStat = fs.statSync(cwd)
1630
+ } catch {
1631
+ cwdStat = undefined
1632
+ }
1633
+ if (cwdStat === undefined || !cwdStat.isDirectory()) {
1634
+ const rel = path.relative(ctx.workspaceRoot, cwd).toPosix() || path.basename(cwd)
1635
+ return err(
1636
+ `execute_command: cwd '${rel}' does not exist in this workspace.\n\nRecovery suggestions:\n1. Use list_files to confirm the real directory structure before setting cwd\n2. Omit cwd (or pass null) to run in the workspace root\n3. If you meant a path from a different task/workspace, it doesn't exist here`,
1637
+ )
1638
+ }
1571
1639
  }
1572
1640
 
1573
1641
  // Permissions gate (command allow/deny + dangerous substitution +
@@ -1889,7 +1957,15 @@ function listFilesHandler(
1889
1957
  const key = `${target} ${recursive}`
1890
1958
  const entry = calls?.get(key)
1891
1959
 
1892
- if (calls !== undefined && generation !== undefined && entry !== undefined && entry.generation === generation) {
1960
+ // See readFileHandler's disableReadFileCache comment for the full
1961
+ // rationale (same short-circuit shape, same local-backend risk).
1962
+ if (
1963
+ !ctx.disableReadFileCache &&
1964
+ calls !== undefined &&
1965
+ generation !== undefined &&
1966
+ entry !== undefined &&
1967
+ entry.generation === generation
1968
+ ) {
1893
1969
  if (!entry.toldUnchanged) {
1894
1970
  entry.toldUnchanged = true
1895
1971
  return ok(