headlesscode 1.0.2 → 1.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/tools/executor.ts +153 -4
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "headlesscode",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.3",
|
|
4
4
|
"description": "Standalone headless coding-agent harness: runs the Zoo Code agent loop (prompts, tools, modes) without a VS Code UI, driven by CLI, HTTP, and parallel worktree orchestration.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"repository": {
|
package/src/tools/executor.ts
CHANGED
|
@@ -1695,6 +1695,7 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
|
|
|
1695
1695
|
env: { ...process.env, LANG: "en_US.UTF-8", LC_ALL: "en_US.UTF-8" },
|
|
1696
1696
|
})
|
|
1697
1697
|
} catch (error) {
|
|
1698
|
+
logSpawnDiagnostics(command, attempt, "sync-catch", error)
|
|
1698
1699
|
if (attempt < MAX_SPAWN_ATTEMPTS && isBashSpawnEnoent(error)) {
|
|
1699
1700
|
// A live standalone repro of one of these exact failures spawned
|
|
1700
1701
|
// cleanly on the first try outside the long-running harness
|
|
@@ -1757,6 +1758,7 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
|
|
|
1757
1758
|
attemptSettled = true
|
|
1758
1759
|
clearTimeout(timer)
|
|
1759
1760
|
backgroundCommands.delete(child)
|
|
1761
|
+
logSpawnDiagnostics(command, attempt, "error-event", error)
|
|
1760
1762
|
if (attempt < MAX_SPAWN_ATTEMPTS && isBashSpawnEnoent(error)) {
|
|
1761
1763
|
stdout = ""
|
|
1762
1764
|
stderr = ""
|
|
@@ -1787,11 +1789,17 @@ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext):
|
|
|
1787
1789
|
// no other real scenario (a genuine command exit code is always
|
|
1788
1790
|
// 0-255), so treat any negative code on the first attempt the same
|
|
1789
1791
|
// way: retry once before surfacing anything to the model.
|
|
1790
|
-
if (
|
|
1791
|
-
|
|
1792
|
-
|
|
1792
|
+
if (code !== null && code < 0) {
|
|
1793
|
+
// The same bash-spawn-ENOENT failure surfaced via `close`
|
|
1794
|
+
// (no `error` event). Snapshot system state at the moment of
|
|
1795
|
+
// the failure, before any retry backoff begins (issue #1).
|
|
1796
|
+
logSpawnDiagnostics(command, attempt, "close-negative-code", undefined, code)
|
|
1797
|
+
if (attempt < MAX_SPAWN_ATTEMPTS) {
|
|
1798
|
+
stdout = ""
|
|
1799
|
+
stderr = ""
|
|
1793
1800
|
setTimeout(() => spawnAttempt(attempt + 1), SPAWN_RETRY_DELAY_MS * attempt)
|
|
1794
|
-
|
|
1801
|
+
return
|
|
1802
|
+
}
|
|
1795
1803
|
}
|
|
1796
1804
|
if (signal === "SIGKILL") {
|
|
1797
1805
|
// Defensive path only: a timeout never sends SIGKILL anymore,
|
|
@@ -2330,6 +2338,147 @@ function errorMessage(error: unknown): string {
|
|
|
2330
2338
|
return error instanceof Error ? error.message : String(error)
|
|
2331
2339
|
}
|
|
2332
2340
|
|
|
2341
|
+
/**
|
|
2342
|
+
* Spawn-failure diagnostics snapshot (issue #1).
|
|
2343
|
+
*
|
|
2344
|
+
* The bash-spawn-ENOENT failure ("spawn /bin/bash ENOENT", and the
|
|
2345
|
+
* close-event variant with a negative code) has one CONFIRMED contributing
|
|
2346
|
+
* factor — system memory commit pressure: when /proc/meminfo's Committed_AS
|
|
2347
|
+
* sits above ~95-98% of CommitLimit, spawn failure rates spike — plus several
|
|
2348
|
+
* unconfirmed candidates (open-fd exhaustion, the ulimit -u process-count
|
|
2349
|
+
* ceiling, zombie/defunct accumulation, RSS growth in the long-running
|
|
2350
|
+
* harness process). Rather than infer the cause after the fact from separate
|
|
2351
|
+
* /proc snapshots, capture a one-shot snapshot AT the moment of each spawn
|
|
2352
|
+
* failure. Every read is best-effort: a missing /proc file (non-Linux host,
|
|
2353
|
+
* restricted container) yields "n/a" for that field instead of throwing, and
|
|
2354
|
+
* a failure here must never abort the spawn-retry logic this is diagnosing.
|
|
2355
|
+
*/
|
|
2356
|
+
export function collectSpawnDiagnostics(): Record<string, string> {
|
|
2357
|
+
const diag: Record<string, string> = {}
|
|
2358
|
+
|
|
2359
|
+
// Memory commit pressure — the confirmed contributor. Values in kB.
|
|
2360
|
+
let committedAsKb = "n/a"
|
|
2361
|
+
let commitLimitKb = "n/a"
|
|
2362
|
+
try {
|
|
2363
|
+
const meminfo = fs.readFileSync("/proc/meminfo", "utf8")
|
|
2364
|
+
for (const line of meminfo.split("\n")) {
|
|
2365
|
+
if (line.startsWith("Committed_AS:")) {
|
|
2366
|
+
committedAsKb = (line.split(/\s+/)[1] ?? "n/a").trim()
|
|
2367
|
+
} else if (line.startsWith("CommitLimit:")) {
|
|
2368
|
+
commitLimitKb = (line.split(/\s+/)[1] ?? "n/a").trim()
|
|
2369
|
+
}
|
|
2370
|
+
}
|
|
2371
|
+
} catch {
|
|
2372
|
+
// non-Linux or /proc not mounted — fields stay "n/a"
|
|
2373
|
+
}
|
|
2374
|
+
diag.committedAsKb = committedAsKb
|
|
2375
|
+
diag.commitLimitKb = commitLimitKb
|
|
2376
|
+
const committed = Number(committedAsKb)
|
|
2377
|
+
const limit = Number(commitLimitKb)
|
|
2378
|
+
diag.committedPct =
|
|
2379
|
+
Number.isFinite(committed) && Number.isFinite(limit) && limit > 0
|
|
2380
|
+
? `${((committed / limit) * 100).toFixed(1)}%`
|
|
2381
|
+
: "n/a"
|
|
2382
|
+
|
|
2383
|
+
// Open fd count for THIS process — fd exhaustion near the soft limit is a
|
|
2384
|
+
// candidate contributor (posix_spawn needs fds for the child's stdio).
|
|
2385
|
+
try {
|
|
2386
|
+
diag.openFds = String(fs.readdirSync("/proc/self/fd").length)
|
|
2387
|
+
} catch {
|
|
2388
|
+
diag.openFds = "n/a"
|
|
2389
|
+
}
|
|
2390
|
+
|
|
2391
|
+
// System process count + zombie/defunct count + the ulimit -u ceiling.
|
|
2392
|
+
// The zombie scan is bounded to the first 1024 pids so the diagnostic
|
|
2393
|
+
// stays lightweight even on a box with tens of thousands of processes —
|
|
2394
|
+
// this runs ON a spawn-failure path and must never make it slower.
|
|
2395
|
+
let procs = 0
|
|
2396
|
+
let zombies = 0
|
|
2397
|
+
try {
|
|
2398
|
+
const pids = fs.readdirSync("/proc").filter((e) => /^\d+$/.test(e))
|
|
2399
|
+
procs = pids.length
|
|
2400
|
+
for (const pid of pids.slice(0, 1024)) {
|
|
2401
|
+
try {
|
|
2402
|
+
const stat = fs.readFileSync(`/proc/${pid}/stat`, "utf8")
|
|
2403
|
+
// comm can contain spaces and parens — the state char is the
|
|
2404
|
+
// first field after the LAST ')', two chars later.
|
|
2405
|
+
const closeParen = stat.lastIndexOf(")")
|
|
2406
|
+
if (closeParen !== -1 && closeParen + 2 < stat.length && stat[closeParen + 2] === "Z") {
|
|
2407
|
+
zombies++
|
|
2408
|
+
}
|
|
2409
|
+
} catch {
|
|
2410
|
+
// pid exited between readdir and read — not a zombie
|
|
2411
|
+
}
|
|
2412
|
+
}
|
|
2413
|
+
} catch {
|
|
2414
|
+
// non-Linux — both stay at their defaults
|
|
2415
|
+
}
|
|
2416
|
+
diag.systemProcs = procs > 0 ? String(procs) : "n/a"
|
|
2417
|
+
diag.zombies = procs > 0 ? String(zombies) : "n/a"
|
|
2418
|
+
let maxProcs = "n/a"
|
|
2419
|
+
try {
|
|
2420
|
+
const limits = fs.readFileSync("/proc/self/limits", "utf8")
|
|
2421
|
+
const m = limits.match(/Max processes\s+(\d+)/)
|
|
2422
|
+
if (m) {
|
|
2423
|
+
maxProcs = m[1]!
|
|
2424
|
+
}
|
|
2425
|
+
} catch {
|
|
2426
|
+
// non-Linux
|
|
2427
|
+
}
|
|
2428
|
+
diag.maxProcs = maxProcs
|
|
2429
|
+
|
|
2430
|
+
// The long-running harness's own footprint — RSS climbs slowly over hours
|
|
2431
|
+
// (the observed llama-server pattern); heap/uptime come from process.*.
|
|
2432
|
+
let rssMb = "n/a"
|
|
2433
|
+
try {
|
|
2434
|
+
const status = fs.readFileSync("/proc/self/status", "utf8")
|
|
2435
|
+
const m = status.match(/VmRSS:\s+(\d+) kB/)
|
|
2436
|
+
if (m) {
|
|
2437
|
+
rssMb = `${(Number(m[1]!) / 1024).toFixed(0)}`
|
|
2438
|
+
}
|
|
2439
|
+
} catch {
|
|
2440
|
+
// non-Linux
|
|
2441
|
+
}
|
|
2442
|
+
diag.rssMb = rssMb
|
|
2443
|
+
diag.heapMb = `${Math.round(process.memoryUsage().heapUsed / (1024 * 1024))}`
|
|
2444
|
+
diag.uptimeS = `${Math.round(process.uptime())}`
|
|
2445
|
+
|
|
2446
|
+
return diag
|
|
2447
|
+
}
|
|
2448
|
+
|
|
2449
|
+
/** Render a diagnostics snapshot as one grep-friendly `key=value` line. */
|
|
2450
|
+
export function formatSpawnDiagnostics(diag: Record<string, string>): string {
|
|
2451
|
+
return [
|
|
2452
|
+
`committed=${diag.committedPct} (${diag.committedAsKb}/${diag.commitLimitKb} kB)`,
|
|
2453
|
+
`openFds=${diag.openFds}`,
|
|
2454
|
+
`procs=${diag.systemProcs}`,
|
|
2455
|
+
`zombies=${diag.zombies}`,
|
|
2456
|
+
`maxProcs=${diag.maxProcs}`,
|
|
2457
|
+
`rss=${diag.rssMb}MB`,
|
|
2458
|
+
`heap=${diag.heapMb}MB`,
|
|
2459
|
+
`uptime=${diag.uptimeS}s`,
|
|
2460
|
+
].join(" ")
|
|
2461
|
+
}
|
|
2462
|
+
|
|
2463
|
+
/**
|
|
2464
|
+
* One-line, always-on stderr log emitted at the moment of a spawn failure.
|
|
2465
|
+
* Unconditional (NOT HEADLESSCODE_DEBUG-gated): the whole point is that the
|
|
2466
|
+
* NEXT real-session occurrence gets captured without anyone having to
|
|
2467
|
+
* remember to enable a flag first.
|
|
2468
|
+
*/
|
|
2469
|
+
function logSpawnDiagnostics(command: string, attempt: number, shape: string, error?: unknown, code?: number | null): void {
|
|
2470
|
+
const detail =
|
|
2471
|
+
error !== undefined
|
|
2472
|
+
? ` error=${errorMessage(error)}`
|
|
2473
|
+
: code !== undefined
|
|
2474
|
+
? ` code=${code}`
|
|
2475
|
+
: ""
|
|
2476
|
+
const shown = command.length > 200 ? `${command.slice(0, 200)}…` : command
|
|
2477
|
+
process.stderr.write(
|
|
2478
|
+
`[execute_command spawn-diagnostics] shape=${shape} attempt=${attempt}${detail} cmd=${JSON.stringify(shown)} ${formatSpawnDiagnostics(collectSpawnDiagnostics())}\n`,
|
|
2479
|
+
)
|
|
2480
|
+
}
|
|
2481
|
+
|
|
2333
2482
|
/**
|
|
2334
2483
|
* Lazily construct the session's local summarizer. Returns undefined when the
|
|
2335
2484
|
* feature is off (never construct an Ollama client for a non-opted-in
|