headlesscode 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -190,6 +190,8 @@ export interface ToolExecutorOptions {
190
190
  permissions?: PermissionsConfig
191
191
  /** See ToolContext.guardLargeOverwrites (types.ts) for the full writeup. */
192
192
  guardLargeOverwrites?: boolean
193
+ /** See ToolContext.disableReadFileCache (types.ts) for the full writeup. */
194
+ disableReadFileCache?: boolean
193
195
  /**
194
196
  * Live worker monitoring: fired at the same lifecycle points where the
195
197
  * `.harness.needs-decision` marker is written/cleared, so the session can
@@ -359,6 +361,7 @@ export class ToolExecutor {
359
361
  workspaceRoot: this.workspaceRoot,
360
362
  permissions: this.permissions,
361
363
  guardLargeOverwrites: this.options.guardLargeOverwrites,
364
+ disableReadFileCache: this.options.disableReadFileCache,
362
365
  decisionTimeoutMs: this.options.decisionTimeoutMs,
363
366
  decisionPollIntervalMs: this.options.decisionPollIntervalMs,
364
367
  pauseBudgetClock: this.pauseBudgetClock,
@@ -799,7 +802,32 @@ function readFileHandler(
799
802
  // cached hash can be reused without re-reading the file; any mismatch
800
803
  // (including a same-length rewrite, which changes mtime) falls back to
801
804
  // a full read + sha256 below.
805
+ //
806
+ // 2026-09-02: real, confirmed, live-observed failure mode of this
807
+ // short-circuit against a local model -- ctx.disableReadFileCache
808
+ // skips both cache-hit checks in this function entirely, always
809
+ // serving real content. A session's edit_file call failed ("no
810
+ // match found"), the error told it to re-read and retry, it DID
811
+ // call read_file again exactly as instructed, and got back
812
+ // "[cache] this file is unchanged... re-read the earlier tool
813
+ // result" instead of the actual content -- correct per this
814
+ // mechanism's own design (the safety valve is "the SECOND
815
+ // consecutive identical call serves real content again"), but the
816
+ // model never made that second call: it read the cache-hit message
817
+ // as "you already have what you need", gave up, and called
818
+ // attempt_completion claiming the endpoint worked -- a genuine
819
+ // fabrication directly caused by this response, not a model
820
+ // hallucination from nothing. This short-circuit's own rationale
821
+ // (avoid paying full token cost for content the conversation
822
+ // already has) is a real concern for a REMOTE model's per-token API
823
+ // bill; re-serving a few hundred lines of file content costs a
824
+ // local session near-nothing (it's prefill, not generation, so it
825
+ // barely affects wall-clock time either) against a GPU with no
826
+ // per-token price. Cheap insurance against a much more expensive
827
+ // failure mode locally; the cloud sessions this genuinely saves
828
+ // money for are unaffected (disableReadFileCache stays unset there).
802
829
  if (
830
+ !ctx.disableReadFileCache &&
803
831
  cache !== undefined &&
804
832
  !cache.toldUnchanged &&
805
833
  cache.size === stat.size &&
@@ -818,8 +846,9 @@ function readFileHandler(
818
846
 
819
847
  // Cache-check the CURRENT on-disk content (never "no write tool was
820
848
  // called"): identical args + identical hash => byte-identical output.
849
+ // See the disableReadFileCache comment on the check above.
821
850
  const currentHash = hashFileContent(content)
822
- if (cache !== undefined && cache.hash === currentHash && !cache.toldUnchanged) {
851
+ if (!ctx.disableReadFileCache && cache !== undefined && cache.hash === currentHash && !cache.toldUnchanged) {
823
852
  cache.toldUnchanged = true
824
853
  return ok(READ_FILE_CACHE_HIT_MESSAGE)
825
854
  }
@@ -1565,9 +1594,48 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
1565
1594
 
1566
1595
  return (async () => {
1567
1596
  let cwd = ctx.workspaceRoot
1568
- if (args.cwd != null && args.cwd !== "") {
1597
+ // 2026-09-02: real, confirmed, fully deterministic bug -- a local
1598
+ // LoRA backend's tool-call parser was found sending the literal
1599
+ // 4-character STRING "null" (or "None") for an unset optional
1600
+ // `cwd`, not a real absent/null value. `args.cwd != null` is only
1601
+ // false for the JS primitive null -- a non-empty string "null"
1602
+ // sails through this check as if it were a real requested cwd,
1603
+ // resolving to "<workspaceRoot>/null", which (almost) never
1604
+ // exists. That was fixed at the source (the LoRA server's own
1605
+ // parser), but defend here too: any backend/model can make the
1606
+ // same JSON-serialization slip, and treating the literal text
1607
+ // null/None the same as a real null is cheap and unambiguous --
1608
+ // no real directory is ever named exactly "null" or "None".
1609
+ const isNullish = args.cwd == null || args.cwd === "" || args.cwd === "null" || args.cwd === "None"
1610
+ if (!isNullish) {
1569
1611
  const cwdArg = requireString(args, "cwd")
1570
1612
  cwd = safeTarget(ctx, cwdArg)
1613
+ // Validate BEFORE spawning rather than let a bad cwd reach
1614
+ // spawn(): Node's child_process misattributes a chdir()
1615
+ // failure (nonexistent cwd) to the SHELL itself -- "spawn
1616
+ // /bin/bash ENOENT" -- with no mention of cwd anywhere in the
1617
+ // error, which is exactly the "infrastructure spawn failure"
1618
+ // pattern several ground-truth eval runs hit today (verified
1619
+ // directly: spawning with a nonexistent cwd reproduces that
1620
+ // precise error string/code/path). That misattribution isn't
1621
+ // specific to the null-string bug above -- ANY nonexistent
1622
+ // cwd the model supplies (a stale/hallucinated path) hits it
1623
+ // the same way. Checking here turns a misleading spawn crash
1624
+ // into a clear, actionable, model-facing error instead, and
1625
+ // correctly counts it as a real mistake (it's the model's
1626
+ // bad path, not infrastructure).
1627
+ let cwdStat: fs.Stats | undefined
1628
+ try {
1629
+ cwdStat = fs.statSync(cwd)
1630
+ } catch {
1631
+ cwdStat = undefined
1632
+ }
1633
+ if (cwdStat === undefined || !cwdStat.isDirectory()) {
1634
+ const rel = path.relative(ctx.workspaceRoot, cwd).toPosix() || path.basename(cwd)
1635
+ return err(
1636
+ `execute_command: cwd '${rel}' does not exist in this workspace.\n\nRecovery suggestions:\n1. Use list_files to confirm the real directory structure before setting cwd\n2. Omit cwd (or pass null) to run in the workspace root\n3. If you meant a path from a different task/workspace, it doesn't exist here`,
1637
+ )
1638
+ }
1571
1639
  }
1572
1640
 
1573
1641
  // Permissions gate (command allow/deny + dangerous substitution +
@@ -1695,6 +1763,7 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
1695
1763
  env: { ...process.env, LANG: "en_US.UTF-8", LC_ALL: "en_US.UTF-8" },
1696
1764
  })
1697
1765
  } catch (error) {
1766
+ logSpawnDiagnostics(command, attempt, "sync-catch", error)
1698
1767
  if (attempt < MAX_SPAWN_ATTEMPTS && isBashSpawnEnoent(error)) {
1699
1768
  // A live standalone repro of one of these exact failures spawned
1700
1769
  // cleanly on the first try outside the long-running harness
@@ -1757,6 +1826,7 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
1757
1826
  attemptSettled = true
1758
1827
  clearTimeout(timer)
1759
1828
  backgroundCommands.delete(child)
1829
+ logSpawnDiagnostics(command, attempt, "error-event", error)
1760
1830
  if (attempt < MAX_SPAWN_ATTEMPTS && isBashSpawnEnoent(error)) {
1761
1831
  stdout = ""
1762
1832
  stderr = ""
@@ -1787,11 +1857,17 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
1787
1857
  // no other real scenario (a genuine command exit code is always
1788
1858
  // 0-255), so treat any negative code on the first attempt the same
1789
1859
  // way: retry once before surfacing anything to the model.
1790
- if (attempt < MAX_SPAWN_ATTEMPTS && code !== null && code < 0) {
1791
- stdout = ""
1792
- stderr = ""
1860
+ if (code !== null && code < 0) {
1861
+ // The same bash-spawn-ENOENT failure surfaced via `close`
1862
+ // (no `error` event). Snapshot system state at the moment of
1863
+ // the failure, before any retry backoff begins (issue #1).
1864
+ logSpawnDiagnostics(command, attempt, "close-negative-code", undefined, code)
1865
+ if (attempt < MAX_SPAWN_ATTEMPTS) {
1866
+ stdout = ""
1867
+ stderr = ""
1793
1868
  setTimeout(() => spawnAttempt(attempt + 1), SPAWN_RETRY_DELAY_MS * attempt)
1794
- return
1869
+ return
1870
+ }
1795
1871
  }
1796
1872
  if (signal === "SIGKILL") {
1797
1873
  // Defensive path only: a timeout never sends SIGKILL anymore,
@@ -1881,7 +1957,15 @@ function listFilesHandler(
1881
1957
  const key = `${target} ${recursive}`
1882
1958
  const entry = calls?.get(key)
1883
1959
 
1884
- if (calls !== undefined && generation !== undefined && entry !== undefined && entry.generation === generation) {
1960
+ // See readFileHandler's disableReadFileCache comment for the full
1961
+ // rationale (same short-circuit shape, same local-backend risk).
1962
+ if (
1963
+ !ctx.disableReadFileCache &&
1964
+ calls !== undefined &&
1965
+ generation !== undefined &&
1966
+ entry !== undefined &&
1967
+ entry.generation === generation
1968
+ ) {
1885
1969
  if (!entry.toldUnchanged) {
1886
1970
  entry.toldUnchanged = true
1887
1971
  return ok(
@@ -2330,6 +2414,147 @@ function errorMessage(error: unknown): string {
2330
2414
  return error instanceof Error ? error.message : String(error)
2331
2415
  }
2332
2416
 
2417
+ /**
2418
+ * Spawn-failure diagnostics snapshot (issue #1).
2419
+ *
2420
+ * The bash-spawn-ENOENT failure ("spawn /bin/bash ENOENT", and the
2421
+ * close-event variant with a negative code) has one CONFIRMED contributing
2422
+ * factor — system memory commit pressure: when /proc/meminfo's Committed_AS
2423
+ * sits above ~95-98% of CommitLimit, spawn failure rates spike — plus several
2424
+ * unconfirmed candidates (open-fd exhaustion, the ulimit -u process-count
2425
+ * ceiling, zombie/defunct accumulation, RSS growth in the long-running
2426
+ * harness process). Rather than infer the cause after the fact from separate
2427
+ * /proc snapshots, capture a one-shot snapshot AT the moment of each spawn
2428
+ * failure. Every read is best-effort: a missing /proc file (non-Linux host,
2429
+ * restricted container) yields "n/a" for that field instead of throwing, and
2430
+ * a failure here must never abort the spawn-retry logic this is diagnosing.
2431
+ */
2432
+ export function collectSpawnDiagnostics(): Record<string, string> {
2433
+ const diag: Record<string, string> = {}
2434
+
2435
+ // Memory commit pressure — the confirmed contributor. Values in kB.
2436
+ let committedAsKb = "n/a"
2437
+ let commitLimitKb = "n/a"
2438
+ try {
2439
+ const meminfo = fs.readFileSync("/proc/meminfo", "utf8")
2440
+ for (const line of meminfo.split("\n")) {
2441
+ if (line.startsWith("Committed_AS:")) {
2442
+ committedAsKb = (line.split(/\s+/)[1] ?? "n/a").trim()
2443
+ } else if (line.startsWith("CommitLimit:")) {
2444
+ commitLimitKb = (line.split(/\s+/)[1] ?? "n/a").trim()
2445
+ }
2446
+ }
2447
+ } catch {
2448
+ // non-Linux or /proc not mounted — fields stay "n/a"
2449
+ }
2450
+ diag.committedAsKb = committedAsKb
2451
+ diag.commitLimitKb = commitLimitKb
2452
+ const committed = Number(committedAsKb)
2453
+ const limit = Number(commitLimitKb)
2454
+ diag.committedPct =
2455
+ Number.isFinite(committed) && Number.isFinite(limit) && limit > 0
2456
+ ? `${((committed / limit) * 100).toFixed(1)}%`
2457
+ : "n/a"
2458
+
2459
+ // Open fd count for THIS process — fd exhaustion near the soft limit is a
2460
+ // candidate contributor (posix_spawn needs fds for the child's stdio).
2461
+ try {
2462
+ diag.openFds = String(fs.readdirSync("/proc/self/fd").length)
2463
+ } catch {
2464
+ diag.openFds = "n/a"
2465
+ }
2466
+
2467
+ // System process count + zombie/defunct count + the ulimit -u ceiling.
2468
+ // The zombie scan is bounded to the first 1024 pids so the diagnostic
2469
+ // stays lightweight even on a box with tens of thousands of processes —
2470
+ // this runs ON a spawn-failure path and must never make it slower.
2471
+ let procs = 0
2472
+ let zombies = 0
2473
+ try {
2474
+ const pids = fs.readdirSync("/proc").filter((e) => /^\d+$/.test(e))
2475
+ procs = pids.length
2476
+ for (const pid of pids.slice(0, 1024)) {
2477
+ try {
2478
+ const stat = fs.readFileSync(`/proc/${pid}/stat`, "utf8")
2479
+ // comm can contain spaces and parens — the state char is the
2480
+ // first field after the LAST ')', two chars later.
2481
+ const closeParen = stat.lastIndexOf(")")
2482
+ if (closeParen !== -1 && closeParen + 2 < stat.length && stat[closeParen + 2] === "Z") {
2483
+ zombies++
2484
+ }
2485
+ } catch {
2486
+ // pid exited between readdir and read — not a zombie
2487
+ }
2488
+ }
2489
+ } catch {
2490
+ // non-Linux — both stay at their defaults
2491
+ }
2492
+ diag.systemProcs = procs > 0 ? String(procs) : "n/a"
2493
+ diag.zombies = procs > 0 ? String(zombies) : "n/a"
2494
+ let maxProcs = "n/a"
2495
+ try {
2496
+ const limits = fs.readFileSync("/proc/self/limits", "utf8")
2497
+ const m = limits.match(/Max processes\s+(\d+)/)
2498
+ if (m) {
2499
+ maxProcs = m[1]!
2500
+ }
2501
+ } catch {
2502
+ // non-Linux
2503
+ }
2504
+ diag.maxProcs = maxProcs
2505
+
2506
+ // The long-running harness's own footprint — RSS climbs slowly over hours
2507
+ // (the observed llama-server pattern); heap/uptime come from process.*.
2508
+ let rssMb = "n/a"
2509
+ try {
2510
+ const status = fs.readFileSync("/proc/self/status", "utf8")
2511
+ const m = status.match(/VmRSS:\s+(\d+) kB/)
2512
+ if (m) {
2513
+ rssMb = `${(Number(m[1]!) / 1024).toFixed(0)}`
2514
+ }
2515
+ } catch {
2516
+ // non-Linux
2517
+ }
2518
+ diag.rssMb = rssMb
2519
+ diag.heapMb = `${Math.round(process.memoryUsage().heapUsed / (1024 * 1024))}`
2520
+ diag.uptimeS = `${Math.round(process.uptime())}`
2521
+
2522
+ return diag
2523
+ }
2524
+
2525
+ /** Render a diagnostics snapshot as one grep-friendly `key=value` line. */
2526
+ export function formatSpawnDiagnostics(diag: Record<string, string>): string {
2527
+ return [
2528
+ `committed=${diag.committedPct} (${diag.committedAsKb}/${diag.commitLimitKb} kB)`,
2529
+ `openFds=${diag.openFds}`,
2530
+ `procs=${diag.systemProcs}`,
2531
+ `zombies=${diag.zombies}`,
2532
+ `maxProcs=${diag.maxProcs}`,
2533
+ `rss=${diag.rssMb}MB`,
2534
+ `heap=${diag.heapMb}MB`,
2535
+ `uptime=${diag.uptimeS}s`,
2536
+ ].join(" ")
2537
+ }
2538
+
2539
+ /**
2540
+ * One-line, always-on stderr log emitted at the moment of a spawn failure.
2541
+ * Unconditional (NOT HEADLESSCODE_DEBUG-gated): the whole point is that the
2542
+ * NEXT real-session occurrence gets captured without anyone having to
2543
+ * remember to enable a flag first.
2544
+ */
2545
+ function logSpawnDiagnostics(command: string, attempt: number, shape: string, error?: unknown, code?: number | null): void {
2546
+ const detail =
2547
+ error !== undefined
2548
+ ? ` error=${errorMessage(error)}`
2549
+ : code !== undefined
2550
+ ? ` code=${code}`
2551
+ : ""
2552
+ const shown = command.length > 200 ? `${command.slice(0, 200)}…` : command
2553
+ process.stderr.write(
2554
+ `[execute_command spawn-diagnostics] shape=${shape} attempt=${attempt}${detail} cmd=${JSON.stringify(shown)} ${formatSpawnDiagnostics(collectSpawnDiagnostics())}\n`,
2555
+ )
2556
+ }
2557
+
2333
2558
  /**
2334
2559
  * Lazily construct the session's local summarizer. Returns undefined when the
2335
2560
  * feature is off (never construct an Ollama client for a non-opted-in