@edgehero/pi-dispatch 1.10.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +29 -2
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +363 -13
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1348 -326
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
@@ -1,6 +1,8 @@
1
- import { lstatSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync } from "node:fs";
1
+ import { chmodSync, chownSync, lstatSync, mkdirSync, readFileSync, readdirSync, renameSync, rmdirSync, rmSync, statSync, writeFileSync } from "node:fs";
2
2
  import { isAbsolute, join, relative } from "node:path";
3
3
  import { sanitizeJobId } from "./run-history.mjs";
4
+ import { scrubCredentials } from "./redact.mjs";
5
+ import { TRANSIENT_READ_ERRORS } from "./runtime-observations.mjs";
4
6
 
5
7
  /**
6
8
  * sandbox-store.mjs -- the host side of a resurrectable sandbox (REQ-RESURRECTABLE-SANDBOX,
@@ -23,7 +25,7 @@ import { sanitizeJobId } from "./run-history.mjs";
23
25
  * authority" and is right about an append-once log file. Here an operator working inside a resurrected
24
26
  * sandbox writes into the directory, so mtime would keep moving and the window would never close --
25
27
  * for exactly the directories most likely to be large.
26
- * - A directory whose sandbox container is RUNNING is skipped. The sweep runs at worker boot, an
28
+ * - A directory whose sandbox container is RUNNING is skipped. The sweep runs at worker boot and on the retention timer, an
27
29
  * operator's shell can outlive a worker restart by design (the container is named outside the
28
30
  * `pi-job-` reaper's filter), and deleting a live bind mount underneath it is a confusing failure
29
31
  * with a boring cause.
@@ -38,7 +40,87 @@ export const SANDBOX_MANIFEST = "manifest.json";
38
40
  const HOUR_MS = 3600000;
39
41
  const DAY_MS = 86400000;
40
42
 
41
- const defaultFs = { lstatSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync };
43
+ /**
44
+ * The reserved name prefix of a TOMBSTONE in the retention root (issue #446): a retained directory the sweep has
45
+ * decided to delete is first renamed to `.reap-<pid>-<now>-<n>` beside it, and only then deleted. Every reader of the
46
+ * root skips these names (`isSandboxTombstone`): the sweep's own listing, `listSandboxes`, the runtime watch, the
47
+ * network sweep's keep set, doctor's count, and `retainJobDir`, which can never produce one (`sandboxEntryName`).
48
+ *
49
+ * The name carries NO job id, on purpose: `sanitizeJobId` output is unbounded, and a prefix plus a 250-byte id is
50
+ * past NAME_MAX on every common filesystem, so a rename to it would fail with ENAMETOOLONG and hold that run forever.
51
+ */
52
+ export const SANDBOX_TOMBSTONE_PREFIX = ".reap-";
53
+
54
+ /** Whether a name in the retention root is a tombstone rather than a retained run. */
55
+ export function isSandboxTombstone(name) {
56
+ return typeof name === "string" && name.startsWith(SANDBOX_TOMBSTONE_PREFIX);
57
+ }
58
+
59
+ /**
60
+ * How old a tombstone must be before doctor calls it STUCK rather than a delete in progress (issue #446). A tombstone
61
+ * normally lives for exactly one `rmSync`: seconds for a large clone (200k files took about 10 s, measured under
62
+ * #446). One still there ten minutes later is a tree the worker cannot delete, which is not going to resolve itself.
63
+ */
64
+ export const SANDBOX_TOMBSTONE_STUCK_MS = 10 * 60 * 1000;
65
+
66
+ /**
67
+ * How often the sweep repeats its line for the SAME stuck tombstone (issue #446). Each pass retries the delete, and a
68
+ * root-owned file under a non-root worker fails it on every pass, so the line is said when first seen and then once a
69
+ * day per tombstone rather than on every tick; doctor names it in between.
70
+ */
71
+ export const SANDBOX_TOMBSTONE_RELOG_MS = DAY_MS;
72
+
73
+ /** `kill(pid, 0)`: true unless the process is gone (ESRCH). EPERM means alive under another account. */
74
+ export function defaultPidAlive(pid) {
75
+ try {
76
+ process.kill(pid, 0);
77
+ return true;
78
+ } catch (err) {
79
+ return err?.code !== "ESRCH";
80
+ }
81
+ }
82
+
83
+ /** The pid a tombstone's name records, or null when the name is not one this module writes. */
84
+ export function sandboxTombstonePid(name) {
85
+ const m = /^\.reap-(\d+)-\d+-\d+$/.exec(String(name ?? ""));
86
+ return m ? Number(m[1]) : null;
87
+ }
88
+
89
+ /** A tombstone's age at `at` from its name, or null when the name is not one this module writes. */
90
+ export function sandboxTombstoneAge(name, at) {
91
+ const m = /^\.reap-\d+-(\d+)-\d+$/.exec(String(name ?? ""));
92
+ if (!m) return null;
93
+ const made = Number(m[1]);
94
+ return Number.isSafeInteger(made) ? at - made : null;
95
+ }
96
+
97
+ /**
98
+ * The directory name one job id is retained under (issue #446): `sanitizeJobId`, then an ESCAPE for a leading `.` or
99
+ * `_`: one `_` is prepended to either, and nothing else changes. So `.x` is `_.x`, `_x` is `__x`, and `x` is `x`.
100
+ *
101
+ * THE RULE IS PINNED, and it is wider than the tombstone prefix on purpose. `sanitizeJobId` keeps `.`, so an id whose
102
+ * sanitized form is `.reap-...` would name (or be read as) a tombstone, and `.` or `..` would name the retention root
103
+ * itself or its parent, which `retainJobDir` then `rm -rf`s as "the previous attempt" and `pi-dispatch sandbox ..`
104
+ * would read a manifest from. No mapped name starts with `.`, so the whole dot namespace of the root is this module's.
105
+ *
106
+ * AN ESCAPE, NOT A SUBSTITUTION (gate round 1): the first version mapped a leading `.` to `_`, and `sanitizeJobId`
107
+ * emits `_` too, so `.x` and `_x` shared one directory and retaining one deleted the other. Every character outside
108
+ * `sanitizeJobId`'s own set (`[A-Za-z0-9._-]`) is also illegal in a container name, which this name becomes, so the
109
+ * escape stays inside the set and is made unambiguous instead: a mapped name starting with `_` always had one prepended,
110
+ * and one that does not was never changed, so two ids with different sanitized forms never share a name. (Two ids with
111
+ * the SAME sanitized form, `a:b` and `a_b`, still do; that is `sanitizeJobId`'s, and `resolveSandbox` refuses a
112
+ * manifest whose `jobId` is not the one asked for.) No job id this project mints starts with either character.
113
+ *
114
+ * Used by every place that turns an id into this root's name: the retention, the read, the container name
115
+ * (`sandboxContainerName`) and the running checks, so the id the runtime reports and the name the sweep holds stay the
116
+ * same string.
117
+ */
118
+ export function sandboxEntryName(jobId) {
119
+ const name = sanitizeJobId(jobId);
120
+ return name.startsWith(".") || name.startsWith("_") ? `_${name}` : name;
121
+ }
122
+
123
+ const defaultFs = { chmodSync, chownSync, lstatSync, mkdirSync, readFileSync, readdirSync, renameSync, rmdirSync, rmSync, statSync, writeFileSync };
42
124
 
43
125
  /**
44
126
  * Retain one finished job's directory, or delete it.
@@ -47,11 +129,12 @@ const defaultFs = { lstatSync, mkdirSync, readFileSync, readdirSync, renameSync,
47
129
  * nothing was retained -- and on `null` the caller has nothing left to do, because every failure path
48
130
  * here removes `jobDir` itself. Retention must never leave debris behind.
49
131
  *
50
- * `prepared.sandbox` is `{ jobId, kind, image }`, stamped by `makePrepareWorkspace`. Absent (a bare
51
- * construction, a test, an unwired dispatcher) means no retention, which keeps such a caller on exactly
132
+ * `prepared.sandbox` is `{ jobId, kind, image, backend, jobUser? }`, stamped by `makePrepareWorkspace` (`jobUser`
133
+ * only when the processor decided one, issue #341). Absent (a bare construction, a test, an unwired dispatcher)
134
+ * means no retention, which keeps such a caller on exactly
52
135
  * the pre-feature path.
53
136
  */
54
- export function retainJobDir(prepared, { sandboxDir, fs = defaultFs, log = () => {}, now = () => Date.now() } = {}) {
137
+ export function retainJobDir(prepared, { sandboxDir, retentionHours = null, fs = defaultFs, log = () => {}, now = () => Date.now(), euid = process.geteuid?.() } = {}) {
55
138
  const jobDir = prepared?.jobDir;
56
139
  const meta = prepared?.sandbox;
57
140
  if (!jobDir) return null;
@@ -60,7 +143,8 @@ export function retainJobDir(prepared, { sandboxDir, fs = defaultFs, log = () =>
60
143
  return null;
61
144
  }
62
145
 
63
- const dest = join(sandboxDir, sanitizeJobId(meta.jobId));
146
+ const dest = join(sandboxDir, sandboxEntryName(meta.jobId));
147
+ const created = now();
64
148
  try {
65
149
  // FIRST, and load-bearing rather than hygiene. The per-job transcript copy is the most PII-bearing
66
150
  // artifact this system holds -- tool output, file contents, the agent's own reasoning -- and it
@@ -70,6 +154,14 @@ export function retainJobDir(prepared, { sandboxDir, fs = defaultFs, log = () =>
70
154
  fs.rmSync(join(jobDir, "session"), { recursive: true, force: true });
71
155
 
72
156
  fs.mkdirSync(sandboxDir, { recursive: true, mode: 0o700 });
157
+ // Issue #464 (gate round 1): this account's, asked again here and not only at boot. A sandbox dir absent at boot is
158
+ // a name another account can create first (a recursive mkdir takes an existing directory silently), and a workspace
159
+ // renamed into a directory that account owns is one it can swap for its own. Refused like any failed retention:
160
+ // the run is deleted, not kept where someone else controls it. A fake fs with no statSync asks nothing.
161
+ if (Number.isInteger(euid) && typeof fs.statSync === "function") {
162
+ const owner = fs.statSync(sandboxDir).uid;
163
+ if (owner !== euid) throw new Error(`${sandboxDir} is owned by uid ${owner}, not by this account (uid ${euid})`);
164
+ }
73
165
  // A BullMQ retry reuses the job id, so the previous attempt may already be sitting at `dest`. Last
74
166
  // attempt wins: it is the one whose workspace matches the run the operator just watched.
75
167
  fs.rmSync(dest, { recursive: true, force: true });
@@ -79,8 +171,26 @@ export function retainJobDir(prepared, { sandboxDir, fs = defaultFs, log = () =>
79
171
  jobId: meta.jobId,
80
172
  kind: meta.kind ?? null,
81
173
  image: meta.image ?? null,
174
+ // The venue the job resolved to (#277), which `resolveSandbox` refuses by when it is not this
175
+ // host's. `?? null`, never a guessed `local`: a stamp with no venue is refused, and only a manifest
176
+ // that predates the key entirely reads as local.
177
+ backend: meta.backend ?? null,
178
+ // Issue #341: the job user the run had (`{ user, home }`, `user` null = the image's own USER), or null when
179
+ // nothing decided one. The sandbox reuses it for IDENTITY only and still checks the daemon itself.
180
+ jobUser: meta.jobUser ?? null,
181
+ // Issue #429: a podman run's container store (`podman info` graphRoot). Written only when known, so every
182
+ // other manifest is the shape it always was. A sandbox refuses to open under another store, and the sweep
183
+ // holds the run while the podman it asks uses another, because that podman's `ps` answers empty (measured).
184
+ ...(typeof meta.podmanStore === "string" ? { podmanStore: meta.podmanStore } : {}),
82
185
  workspace: rebaseWorkspace(prepared.workspace, jobDir, dest),
83
- createdAt: new Date(now()).toISOString(),
186
+ createdAt: new Date(created).toISOString(),
187
+ // Issue #446: the deadline THIS worker's window gives the run, written down, so an opener whose own
188
+ // PI_SANDBOX_RETENTION_HOURS is LARGER than the worker's (docs/sandbox.md says they routinely differ; the panel
189
+ // is the usual case) does not call open a run the worker is about to delete. Every reader takes the EARLIER of
190
+ // this and `createdAt` plus its own window (`sandboxDeadline`), so a window lowered later still applies. `null`
191
+ // only for a caller that does not say (a bare construction, a test), which reads as `createdAt` plus the
192
+ // reader's window alone, as every manifest from before the key does.
193
+ retainUntil: Number.isFinite(retentionHours) && retentionHours > 0 ? new Date(created + retentionHours * HOUR_MS).toISOString() : null,
84
194
  keepUntil: null,
85
195
  };
86
196
  fs.writeFileSync(join(dest, SANDBOX_MANIFEST), `${JSON.stringify(manifest, null, 2)}\n`, { mode: 0o600 });
@@ -125,13 +235,26 @@ function discard(dir, fs) {
125
235
  /** One retained run by (raw) job id, or null when absent, unreadable or not JSON. */
126
236
  export function readManifest({ sandboxDir, jobId, fs = defaultFs }) {
127
237
  if (!sandboxDir || jobId === undefined || jobId === null) return null;
128
- const dir = join(sandboxDir, sanitizeJobId(jobId));
129
- try {
130
- const manifest = JSON.parse(fs.readFileSync(join(dir, SANDBOX_MANIFEST), "utf8"));
131
- return { ...manifest, dir };
132
- } catch {
133
- return null;
134
- }
238
+ const read = (name) => {
239
+ const dir = join(sandboxDir, name);
240
+ try {
241
+ const manifest = JSON.parse(fs.readFileSync(join(dir, SANDBOX_MANIFEST), "utf8"));
242
+ return { ...manifest, dir };
243
+ } catch {
244
+ return null;
245
+ }
246
+ };
247
+ const escaped = sandboxEntryName(jobId);
248
+ const found = read(escaped);
249
+ if (found && (typeof found.jobId !== "string" || found.jobId === String(jobId))) return found;
250
+ // A run retained BEFORE the escape (#446 gate round 2): an id whose safe form starts with `_` was retained under that
251
+ // form, where it is still listed and still swept. Read there when the escaped name holds nothing OR holds another
252
+ // id's run (gate round 3), never under a dot name, and only when the manifest there IS this id's, so the fallback can
253
+ // never hand over another run. Otherwise the escaped read stands, and `resolveSandbox` refuses a mismatch by name.
254
+ const legacy = sanitizeJobId(jobId);
255
+ if (legacy === escaped || legacy.startsWith(".")) return found;
256
+ const old = read(legacy);
257
+ return old && old.jobId === String(jobId) ? old : found;
135
258
  }
136
259
 
137
260
  /**
@@ -150,6 +273,8 @@ export function listSandboxes({ sandboxDir, fs = defaultFs }) {
150
273
  }
151
274
  const out = [];
152
275
  for (const name of names) {
276
+ // A tombstone is a run already decided for deletion (issue #446): not re-openable, so not listed.
277
+ if (isSandboxTombstone(name)) continue;
153
278
  const dir = join(sandboxDir, name);
154
279
  try {
155
280
  const manifest = JSON.parse(fs.readFileSync(join(dir, SANDBOX_MANIFEST), "utf8"));
@@ -168,32 +293,68 @@ export function listSandboxes({ sandboxDir, fs = defaultFs }) {
168
293
  * is how a directory holding a full repository clone per run becomes unbounded, and the acceptance this
169
294
  * feature was written against says retention stays swept.
170
295
  */
171
- export function pinSandbox({ sandboxDir, jobId, pinDays, fs = defaultFs, now = () => Date.now() }) {
296
+ export function pinSandbox({ sandboxDir, jobId, pinDays, fs = defaultFs, now = () => Date.now(), euid = process.geteuid?.() }) {
172
297
  const manifest = readManifest({ sandboxDir, jobId, fs });
173
298
  if (!manifest) return { pinned: false, reason: "absent" };
174
299
  const keepUntil = new Date(now() + pinDays * DAY_MS).toISOString();
175
300
  const { dir, ...body } = manifest;
301
+ // ATOMIC (issue #429 review): a temp file beside it, then a rename over it. `writeFileSync` on the manifest itself
302
+ // truncates first, and for that window every reader (the retention sweep, `--list`, the panel, an open) sees a
303
+ // manifest that does not parse. A rename is all or nothing.
304
+ //
305
+ // AND THE FILE KEEPS ITS IDENTITY (review round 3, reproduced): a rename puts the TEMP file's owner and mode in
306
+ // place, so `sudo -E pi-dispatch sandbox --pin` left a root-owned 0600 manifest the worker could not read, and the
307
+ // worker's sweep deleted the run as manifest-less after the shell exited. So the manifest is stat'ed first and the
308
+ // temp file given its uid, gid and mode before the rename. `chown` only succeeds as root; as anyone else it is
309
+ // EPERM, which is harmless exactly when this process already owns the file (the temp file is then the same owner),
310
+ // and otherwise the pin is refused rather than handing the manifest to another account. The temp file is created
311
+ // with `wx`, so an existing path (a planted symlink, a leftover) is never written through.
312
+ const target = join(dir, SANDBOX_MANIFEST);
313
+ const tmp = join(dir, `.${SANDBOX_MANIFEST}.${process.pid}.${now()}.tmp`);
314
+ let created = false;
176
315
  try {
177
- fs.writeFileSync(join(dir, SANDBOX_MANIFEST), `${JSON.stringify({ ...body, keepUntil }, null, 2)}\n`, { mode: 0o600 });
316
+ const original = fs.lstatSync(target);
317
+ const mode = Number.isInteger(original?.mode) ? original.mode & 0o777 : 0o600;
318
+ fs.writeFileSync(tmp, `${JSON.stringify({ ...body, keepUntil }, null, 2)}\n`, { mode, flag: "wx" });
319
+ created = true;
320
+ if (Number.isInteger(original?.uid) && Number.isInteger(original?.gid)) {
321
+ try {
322
+ fs.chownSync(tmp, original.uid, original.gid);
323
+ } catch (err) {
324
+ if (!(err?.code === "EPERM" && euid === original.uid)) throw err;
325
+ }
326
+ }
327
+ // Explicit, since `writeFileSync`'s mode is filtered through the umask.
328
+ fs.chmodSync(tmp, mode);
329
+ fs.renameSync(tmp, target);
178
330
  return { pinned: true, keepUntil };
179
331
  } catch (err) {
332
+ if (created) {
333
+ try {
334
+ fs.rmSync(tmp, { force: true });
335
+ } catch {
336
+ // nothing left to try
337
+ }
338
+ }
180
339
  return { pinned: false, reason: err?.message ?? "write-failed" };
181
340
  }
182
341
  }
183
342
 
184
343
  /**
185
- * The boot sweep. Fault isolation is the contract, mirroring makeLogReaper and makeReaper: `reapSandboxes`
344
+ * The retention sweep: at boot, and on the timer since issue #292 (`PI_SWEEP_INTERVAL_HOURS`). Fault isolation is the contract, mirroring makeLogReaper and makeReaper: `reapSandboxes`
186
345
  * NEVER throws under any input, and one bad entry cannot abort the rest of the sweep.
187
346
  *
188
347
  * There is NO keep-forever sentinel here, unlike PI_LOG_RETENTION_DAYS and PI_SESSIONS_TTL_DAYS.
189
- * `retentionHours === 0` is the feature being OFF, and it needs no special case: the cutoff becomes `now`,
190
- * so every unpinned directory is already expired and gets swept. Turning retention off therefore also
191
- * cleans up what an earlier setting retained, while an explicit `--pin` still runs to its own deadline --
192
- * an operator's deliberate act outliving a config change is the behaviour worth having.
348
+ * `retentionHours === 0` is the feature being OFF, and it needs no special case: an unpinned run's deadline is the
349
+ * EARLIER of the one its manifest records (`retainUntil`, issue #446) and `createdAt` plus this window
350
+ * (`sandboxDeadline`), so at 0 every unpinned directory is already expired and gets swept. Turning retention off, or
351
+ * down, therefore also cleans up what an earlier setting retained, while an explicit `--pin` still runs to its own
352
+ * deadline -- an operator's deliberate act outliving a config change is the behaviour worth having.
193
353
  *
194
354
  * `listRunning` yields the JOB IDS of live sandboxes -- ids, not container names, so this module needs to
195
355
  * know nothing about how a container is named and the two files stay acyclic. It defaults to none, so an
196
- * unwired reaper still sweeps; start.mjs injects the docker-backed one.
356
+ * unwired reaper still sweeps; start.mjs injects `makeSandboxRuntimeWatch`'s, which answers the ids to HOLD this
357
+ * pass: those each retained run's own runtime reports open, and those whose runtime could not answer (issue #429).
197
358
  */
198
359
  export function makeSandboxReaper({
199
360
  sandboxDir,
@@ -202,48 +363,374 @@ export function makeSandboxReaper({
202
363
  log = () => {},
203
364
  now = () => Date.now(),
204
365
  listRunning = async () => [],
366
+ // Issue #337: the session NETWORKS, swept after the directories and keyed on them. INJECTED rather than
367
+ // imported, and that is not style: `sandbox.mjs` imports `readManifest` from this module, so importing the
368
+ // sweeper here would make a cycle. It also keeps this file docker-free in its own tests, which its header
369
+ // gives as the reason the two modules are acyclic in the first place. Defaults to a real no-op so every
370
+ // existing construction is byte-unchanged.
371
+ sweepNetworks = async () => ({ swept: [], notes: [] }),
372
+ // Issue #446: the pid in a tombstone's name, seamed so a test can name one.
373
+ pid = process.pid,
374
+ // Whether a pid is a live process (gate round 1): `kill(pid, 0)`, where ESRCH is gone and EPERM is alive under
375
+ // another account. Seamed so a test can name a dead one.
376
+ pidAlive = defaultPidAlive,
377
+ // Issue #446, gate round 1: whether ONE run's sandbox is open RIGHT NOW, asked of its runtime immediately before
378
+ // the run is renamed aside. `true` holds it, and so does a throw. Defaults to "no", so an unwired reaper sweeps as
379
+ // it did; start.mjs injects the runtime watch's `isOpen`.
380
+ isOpen = async () => false,
205
381
  }) {
382
+ // Across passes, never reset: two tombstones of one pass share `pid` and `at`, and a pass on the same millisecond as
383
+ // the last must not reuse a name either.
384
+ let seq = 0;
385
+ // The stuck tombstones already said, and when (`SANDBOX_TOMBSTONE_RELOG_MS`). Only rate-limits a log line; losing it
386
+ // on a restart costs one repeated line.
387
+ const stuckSaid = new Map();
206
388
  return async function reapSandboxes() {
207
389
  if (!sandboxDir) return;
208
- let running = new Set();
390
+ let names;
209
391
  try {
210
- running = new Set(await listRunning());
392
+ names = fs.readdirSync(sandboxDir);
211
393
  } catch (err) {
212
- // Could not ask docker. Sweeping blind risks pulling a mount out from under a live shell, so
213
- // skip this sweep entirely: a directory kept one boot too long is the cheaper mistake.
214
- log("sandbox_reaper_skipped", { reason: err?.message ?? "running-lookup-failed" });
215
- return;
394
+ // A root that does not EXIST is not a failed read, it is an empty one, and the difference matters
395
+ // to the network sweep below (issue #337): a host whose sandbox root was never created or was
396
+ // removed by hand is exactly the host most likely to be holding orphaned `pi-sandbox-` networks,
397
+ // and skipping there would mean the sweep never fires on it at all. It is also safe rather than
398
+ // merely convenient: with no root, `resolveSandbox` refuses every run that resolves the SAME
399
+ // root, so no open can be in flight for the sweep to race. An opener computing a DIFFERENT
400
+ // root (its own environment, which `docs/sandbox.md` says routinely differs) is the residual,
401
+ // and it is bounded: the next open just creates the network again, and one in flight is still
402
+ // held by the sweeper's own container look. Any other error (a permission wall, an I/O fault) is a read that
403
+ // failed and still skips the pass.
404
+ if (err?.code !== "ENOENT") {
405
+ log("sandbox_reaper_skipped", { reason: scrubCredentials(err?.message) });
406
+ return;
407
+ }
408
+ names = [];
216
409
  }
217
410
 
218
- let names;
411
+ // LEFTOVER TOMBSTONES FIRST (issue #446): one a crash left between the rename and the delete, or one whose delete
412
+ // failed on an earlier pass. Removed here on every pass, boot and timer alike, and never read as a run: a
413
+ // tombstone's id has already left `keep`, and nothing can open it. Its line is rate-limited, because a tree the
414
+ // worker cannot delete fails the same way on every pass.
415
+ //
416
+ // ONE WORKER PER RETENTION ROOT is the supported configuration (`INT-SANDBOX-CONTRACT`, on `DES-CONCURRENCY-3`'s
417
+ // one worker per daemon). Still, a tombstone ANOTHER process named is left alone while it is younger than
418
+ // `SANDBOX_TOMBSTONE_STUCK_MS`: that process may be between its rename and its read-back, about to put a pinned run
419
+ // back, and deleting it there would lose the run. The name carries the pid, so this costs one comparison. A
420
+ // tombstone this process named is always from an earlier pass (passes never overlap), and one another process left
421
+ // behind is cleared once it is old enough that no pass could still hold it.
422
+ const runs = [];
423
+ const pass = now();
424
+ // The pinned tombstones this pass could not put back, for ONE retry after the main loop (PR #457's final check).
425
+ const heldPinned = [];
426
+ for (const name of names) {
427
+ if (!isSandboxTombstone(name)) {
428
+ runs.push(name);
429
+ continue;
430
+ }
431
+ const owner = sandboxTombstonePid(name);
432
+ const age = sandboxTombstoneAge(name, pass);
433
+ // Only while that process is ALIVE (gate round 1): a worker that crashed between its rename and its delete
434
+ // and was restarted has a new pid, and its tombstone is a plain leftover to clear now, not in ten minutes. A
435
+ // NEGATIVE age (a name stamped by a clock that ran ahead) is not young, it is unknowable, and is treated as
436
+ // old (gate round 2): otherwise a reused pid would hold it forever. The skip is said once a period.
437
+ if (owner !== null && owner !== pid && age !== null && age >= 0 && age < SANDBOX_TOMBSTONE_STUCK_MS && pidAlive(owner)) {
438
+ sayOnce(name, pass, { reason: "tombstone-foreign" });
439
+ continue;
440
+ }
441
+ if ((await clearTombstone(name, pass)) === "pinned-held") heldPinned.push(name);
442
+ }
443
+ names = runs;
444
+
445
+ // ONE READ DECIDES (issue #429, review round 3). Every retained directory's manifest is read ONCE per pass, here,
446
+ // and that one result is what the runtime watch places the run by AND what expiry decides on. Two reads let a
447
+ // pass decide on one and delete on another: a read that failed for the watch (skipping a hold) and succeeded
448
+ // for expiry, or the reverse, deleted a directory under an open sandbox (reproduced twice). So the listing
449
+ // comes first now, then the reads, then the question to the runtimes, handed the reads.
450
+
451
+ const reads = new Map();
452
+ for (const name of names) reads.set(name, readRetained(fs, join(sandboxDir, name)));
453
+
454
+ let running = new Set();
219
455
  try {
220
- names = fs.readdirSync(sandboxDir);
456
+ running = new Set(await listRunning({ names, reads }));
221
457
  } catch (err) {
222
- log("sandbox_reaper_skipped", { reason: err?.message });
458
+ // Could not ask docker. Sweeping blind risks pulling a mount out from under a live shell, so
459
+ // skip this sweep entirely: a directory kept one boot too long is the cheaper mistake.
460
+ log("sandbox_reaper_skipped", { reason: scrubCredentials(err?.message ?? "running-lookup-failed") });
223
461
  return;
224
462
  }
225
463
 
226
464
  const at = now();
227
- const cutoff = at - retentionHours * HOUR_MS;
465
+ // The listing this pass STARTED with, captured before anything is removed. Keying the network sweep on
466
+ // the survivors instead would reopen the race issue #277 withdrew a fix for: an open that passed
467
+ // `resolveSandbox` while its directory existed, whose directory this same pass then expires, would lose
468
+ // its network between `createJobNetwork` and `launch`. No open in progress can be absent from this list,
469
+ // because `resolveSandbox` refuses a job whose directory is gone.
470
+ const keep = new Set(names);
471
+ // The ids whose directory this pass could NOT move out of the way (issue #363, redefined by #446). Such a
472
+ // directory stays on disk under its own name, so its id stays in `keep` on every later pass and its network is
473
+ // never a candidate again. That is CORRECT (a directory that exists is a run that can be re-opened, so its
474
+ // network is wanted) and it was invisible: `sandbox_reaper_skipped` names a directory and nothing on the host
475
+ // named the network it holds. Since #446 that is a failed RENAME (or a non-directory entry that would not go),
476
+ // which a directory's own contents cannot cause. The ordinary #363 shape, root-owned files in a retained forge
477
+ // workspace under a non-root worker, now renames fine and fails the DELETE: it stays a tombstone, its id leaves
478
+ // `keep` from the next pass (so its network is swept), the run can no longer be opened, which is intended, and
479
+ // each pass retries the delete (`clearTombstone`) while doctor names it.
480
+ const blocked = new Set();
228
481
  for (const name of names) {
229
482
  const dir = join(sandboxDir, name);
483
+ // Set ONLY around a removal, so `blocked` means what its name says. The first version added every
484
+ // throw in this loop, which is a different set: a directory that VANISHED between the listing and
485
+ // the lstat is not a directory that could not be removed, and the note it produced asserted the
486
+ // opposite of what had happened -- that the directory stays on disk, so its id stays in `keep` and
487
+ // its network is never a candidate again. It does not stay, and it is.
488
+ let removing = false;
230
489
  try {
231
490
  // lstat: a symlink here resolves on the host, and this tree is agent-written.
232
491
  if (!fs.lstatSync(dir).isDirectory()) {
492
+ removing = true;
233
493
  fs.rmSync(dir, { recursive: true, force: true });
234
494
  log("reaped_sandbox", { entry: name, reason: "not-a-directory" });
235
495
  continue;
236
496
  }
237
497
  if (running.has(name)) continue; // an operator is inside it
238
- const verdict = expiry(dir, fs, at, cutoff);
498
+ // A manifest that could not be read FOR A MOMENT (`TRANSIENT_READ_ERRORS`: EMFILE, EIO and the rest) is
499
+ // HELD this pass, full stop: no runtime answer can release it, because nothing about the run is known,
500
+ // not even which runtime to ask. The next pass reads it again. Said once per directory per pass.
501
+ const read = reads.get(name);
502
+ if (read?.transient) {
503
+ log("sandbox_reaper_skipped", { entry: name, reason: "manifest-unread" });
504
+ continue;
505
+ }
506
+ const verdict = expiry(read, at, retentionHours);
239
507
  if (!verdict.expired) continue;
240
- fs.rmSync(dir, { recursive: true, force: true });
508
+ // AND AGAIN AT THE POINT OF DELETION (final review of round 3, reproduced): the pass's read is taken BEFORE
509
+ // the runtimes are asked, which can take seconds per runtime, so a pin landing meanwhile, or a BullMQ retry
510
+ // replacing the directory with a fresh run, was deleted on the stale read. The one read still PLACES the
511
+ // run; the delete additionally needs a fresh read (lstat first) that is byte-for-byte the same manifest,
512
+ // so it is the same run under the same expiry inputs. A retained manifest carries its `createdAt` to the
513
+ // millisecond, so a retry's fresh run never matches. Anything else holds the directory this pass: a
514
+ // transient fresh read (`manifest-unread`), or a changed, pinned, replaced or newly unreadable one
515
+ // (`manifest-changed`); the next pass reads it again.
516
+ // ASKED AGAIN, FOR THIS RUN, NOW (gate round 1, reproduced). `running` came from a `ps` taken before the
517
+ // pass read anything, and the pass can reach this directory much later (other runtimes' asks, other
518
+ // deletions), after an open's container started. So the run's own runtime is asked once more, here, and a
519
+ // run it reports open, or cannot answer for, is held. What is left is the microseconds from this answer
520
+ // to the rename, which the opener's post-launch look covers.
521
+ let hold = null;
522
+ try {
523
+ if ((await isOpen({ name, read })) === true) hold = "opened-during-pass";
524
+ } catch {
525
+ hold = "runtime-unanswered";
526
+ }
527
+ if (hold) {
528
+ log("sandbox_reaper_skipped", { entry: name, reason: hold });
529
+ continue;
530
+ }
531
+ const fresh = readRetained(fs, dir);
532
+ if (fresh.transient || !sameRead(read, fresh)) {
533
+ log("sandbox_reaper_skipped", { entry: name, reason: fresh.transient ? "manifest-unread" : "manifest-changed" });
534
+ continue;
535
+ }
536
+ // THE TOMBSTONE (issue #446). The fresh read above is microseconds before the decision, but a recursive
537
+ // delete of a large clone takes SECONDS (200k files, about 10 s), and until it reaches `manifest.json` a pin
538
+ // from another process still succeeds and reports `pinned: true` over a directory that is then gone. So
539
+ // the directory is first renamed out of its name, in the same root (one rename, never a copy), and only
540
+ // the tombstone is deleted: from the rename on, every open, pin and read of this run finds nothing, which
541
+ // is the truth. A rename that fails HOLDS the directory under its own name (`rename-failed`), in the
542
+ // family's one line (`OQ-007`).
543
+ const tombName = `${SANDBOX_TOMBSTONE_PREFIX}${pid}-${at}-${seq++}`;
544
+ const tomb = join(sandboxDir, tombName);
545
+ try {
546
+ fs.renameSync(dir, tomb);
547
+ } catch (err) {
548
+ blocked.add(name);
549
+ log("sandbox_reaper_skipped", { entry: name, reason: "rename-failed", code: err?.code ?? null });
550
+ continue;
551
+ }
552
+ // AND READ IT ONCE MORE, THROUGH THE TOMBSTONE. A pin that landed between the fresh read and the rename is
553
+ // the one write the rename cannot see, and it is in this file now: anything but the pass's own read (a
554
+ // changed manifest, or one that cannot be read just now) puts the directory back, when nothing has taken
555
+ // its name meanwhile, and holds it as `manifest-changed`. After this read a pin cannot land: it reads the
556
+ // run's own name, which is gone. The worker's own `retainJobDir` cannot interleave here, since it runs in
557
+ // this same process and this stretch is synchronous.
558
+ const moved = readRetained(fs, tomb);
559
+ if (moved.transient || !sameRead(read, moved)) {
560
+ const restored = restoreTombstone(tomb, dir);
561
+ log("sandbox_reaper_skipped", { entry: name, reason: "manifest-changed", ...(restored ? {} : { restored: false }) });
562
+ continue;
563
+ }
564
+ try {
565
+ fs.rmSync(tomb, { recursive: true, force: true });
566
+ } catch (err) {
567
+ // A tombstone that would not go STAYS one: the run is already unopenable, and the next pass retries.
568
+ stuckSaid.set(tombName, at);
569
+ log("sandbox_reaper_skipped", { entry: name, reason: "tombstone-stuck", tombstone: tombName, code: err?.code ?? null });
570
+ await new Promise((resolve) => setImmediate(resolve));
571
+ continue;
572
+ }
241
573
  log("reaped_sandbox", { entry: name, reason: verdict.reason });
574
+ // Yield after each tree. Free at boot, where nothing is in flight; NOT free since issue
575
+ // #292 put this on a timer beside draining jobs, because a retained directory is a
576
+ // repository clone and `rmSync` is synchronous, so deleting a set of them back to back
577
+ // wedges the event loop for the whole set. `index.mjs` runs with `maxStalledCount: 0`
578
+ // against BullMQ's 30s lock, so a block past the renewal window FAILS a paid job. This
579
+ // bounds the contiguous block to ONE directory, which is the part that cannot be yielded.
580
+ await new Promise((resolve) => setImmediate(resolve));
242
581
  } catch (err) {
243
- log("sandbox_reaper_skipped", { entry: name, reason: err?.message });
582
+ if (removing) blocked.add(name);
583
+ log("sandbox_reaper_skipped", { entry: name, reason: scrubCredentials(err?.message) });
244
584
  }
245
585
  }
586
+
587
+ // THE HELD PINNED TOMBSTONES, ONCE MORE (PR #457's final check). One held because a directory with no manifest held
588
+ // its run's name waited a whole pass for the next leftover sweep, though the loop above deletes such a directory
589
+ // (`no-manifest`) in this same pass. Retried here, after the loop and before the network sweep (whose fresh
590
+ // listing then sees the run back), through the same `clearTombstone`: its reads, its restore that displaces only an
591
+ // empty directory, its rule that never deletes a live pin, and its once-per-period line when the name is still taken.
592
+ for (const name of heldPinned) await clearTombstone(name, pass);
593
+
594
+ // The session networks, after the directories. Reached only when `listRunning` ANSWERED and the
595
+ // directory listing was read, which is the same precondition the directory pass has: without either,
596
+ // nothing here can be called unclaimed.
597
+ //
598
+ // The keep set is the UNION of two listings, and each half covers what the other cannot. The one this
599
+ // pass STARTED with covers a run whose directory this pass then expired: an open that passed
600
+ // `resolveSandbox` a moment before is mid-launch and must not lose its network. A FRESH one covers the
601
+ // opposite end: `retainJobDir` creates a directory at job end, in this same process, and this pass
602
+ // awaits docker and yields per tree, so a job can finish and its run be opened while the pass is still
603
+ // running. That id is in neither the old listing nor `running` -- the container is not up yet -- so
604
+ // without the second read the sweep would take a network `createJobNetwork` had just made.
605
+ //
606
+ // The fresh read is handed over as a CLOSURE rather than as a set, so the sweeper can take it after
607
+ // its own candidate listing: evidence that protects a network must never be older than the listing
608
+ // that nominated it. A read that throws leaves the sweeper and lands in the catch below, which is
609
+ // deliberate -- half a keep set is worse than no sweep. ENOENT is an empty root, not a failure, for
610
+ // the reason given above.
611
+ //
612
+ // Its own fault keeps the `sandbox_reaper_skipped` name on `OQ-007`'s stated property, that one grep
613
+ // covers boot and every tick; only the per-network VERDICTS get new names.
614
+ try {
615
+ const retained = () => {
616
+ try {
617
+ // A tombstone is not a retained run (issue #446), and its name is not an id either.
618
+ return fs.readdirSync(sandboxDir).filter((name) => !isSandboxTombstone(name));
619
+ } catch (err) {
620
+ if (err?.code === "ENOENT") return [];
621
+ throw err;
622
+ }
623
+ };
624
+ const { swept, notes, failed, failures } = await sweepNetworks({ running, keep, retained, blocked });
625
+ for (const s of swept) log("reaped_sandbox_network", s);
626
+ for (const n of notes) log("sandbox_network_not_reaped", n);
627
+ // One line per runtime whose listing failed, naming it (issue #429), where a sweeper that knows its runtime
628
+ // says so; a bare `failed` from a single sweeper keeps the line it always had.
629
+ if (Array.isArray(failures) && failures.length > 0) for (const f of failures) log("sandbox_reaper_skipped", { reason: f.reason, runtime: f.runtime });
630
+ else if (failed) log("sandbox_reaper_skipped", { reason: failed });
631
+ } catch (err) {
632
+ log("sandbox_reaper_skipped", { reason: scrubCredentials(err?.message ?? "network-sweep-failed") });
633
+ }
246
634
  };
635
+
636
+ /**
637
+ * The name a pinned tombstone goes back under, or null to HOLD it: only ever one of the two names its own `jobId`
638
+ * maps to, `sandboxEntryName(jobId)` or the pre-escape `sanitizeJobId(jobId)` (gate round 3: the first version took
639
+ * the first path component of `workspace`, so a crafted manifest could restore a tombstone under ANOTHER run's
640
+ * name). The workspace path only CHOOSES between those two, so a run retained before the escape keeps the name its
641
+ * recorded paths use. No usable `jobId`, no restore.
642
+ */
643
+ function restoreName(manifest) {
644
+ const jobId = manifest?.jobId;
645
+ if (typeof jobId !== "string" || jobId === "") return null;
646
+ const escaped = sandboxEntryName(jobId);
647
+ const legacy = sanitizeJobId(jobId);
648
+ if (legacy === escaped || legacy.startsWith(".")) return escaped;
649
+ const rel = typeof manifest.workspace === "string" ? relative(sandboxDir, manifest.workspace) : "";
650
+ return rel.split(/[\\/]/)[0] === legacy && !rel.startsWith("..") && !isAbsolute(rel) ? legacy : escaped;
651
+ }
652
+
653
+ /** Say `fields` about tombstone `name` when first seen and then once per `SANDBOX_TOMBSTONE_RELOG_MS`. */
654
+ function sayOnce(name, at, fields) {
655
+ const said = stuckSaid.get(name);
656
+ if (said !== undefined && at - said < SANDBOX_TOMBSTONE_RELOG_MS) return;
657
+ stuckSaid.set(name, at);
658
+ log("sandbox_reaper_skipped", { entry: name, ...fields });
659
+ }
660
+
661
+ /**
662
+ * Delete one leftover tombstone; a failure is said when first seen and then once per `SANDBOX_TOMBSTONE_RELOG_MS`.
663
+ *
664
+ * NEVER ONE HOLDING A LIVE PIN (gate round 1, belt and braces). A tombstone is only ever made of an expired run, so a
665
+ * manifest in it with an unexpired `keepUntil` is a pin that landed during the sweep and whose restore did not
666
+ * happen (a crash between the rename and the read-back, or a name that could not be freed). Its pinner was told
667
+ * `pinned: true`. It is restored under its run's name when that is free (or an empty directory), and otherwise held
668
+ * and said (`tombstone-pinned`); once its pin lapses it is an ordinary leftover. A manifest that cannot be read for
669
+ * a moment holds it too: deleting on an unknown is the one mistake here that cannot be undone.
670
+ *
671
+ * Resolves `"pinned-held"` for a pinned one whose run's name was taken, which the pass retries once after its main
672
+ * loop; null otherwise.
673
+ */
674
+ async function clearTombstone(name, at) {
675
+ const path = join(sandboxDir, name);
676
+ const read = readRetained(fs, path);
677
+ const keepUntil = Date.parse(read.manifest?.keepUntil ?? "");
678
+ let held = false;
679
+ if (read.transient) {
680
+ sayOnce(name, at, { reason: "manifest-unread" });
681
+ } else if (Number.isFinite(keepUntil) && keepUntil > at) {
682
+ const target = restoreName(read.manifest);
683
+ const restored = target !== null && restoreTombstone(path, join(sandboxDir, target));
684
+ if (restored) {
685
+ stuckSaid.delete(name);
686
+ log("sandbox_reaper_skipped", { entry: name, reason: "tombstone-pinned", restored: true });
687
+ } else {
688
+ held = target !== null;
689
+ sayOnce(name, at, { reason: "tombstone-pinned", restored: false });
690
+ }
691
+ } else {
692
+ try {
693
+ fs.rmSync(path, { recursive: true, force: true });
694
+ stuckSaid.delete(name);
695
+ log("reaped_sandbox", { entry: name, reason: "tombstone" });
696
+ } catch (err) {
697
+ sayOnce(name, at, { reason: "tombstone-stuck", code: err?.code ?? null });
698
+ }
699
+ }
700
+ // The per-tree yield of the main loop, for its reason: this runs on the timer beside draining jobs.
701
+ await new Promise((resolve) => setImmediate(resolve));
702
+ return held ? "pinned-held" : null;
703
+ }
704
+
705
+ /**
706
+ * Put a tombstone back under a run's name. When the name is free, one rename. When it is TAKEN, only an EMPTY
707
+ * directory may be displaced (`rmdirSync` removes nothing else), and that is the one thing that can appear there in
708
+ * these microseconds: Docker's auto-created bind source, created by an open that lost this race (Podman refuses to bind a
709
+ * missing path instead, exit 125, which the opener reports as swept). Gate round 1
710
+ * reproduced the alternative with a real filesystem: leaving the run as a tombstone because the name was taken
711
+ * handed a PINNED run to the next pass's leftover sweep. The open sitting on the displaced empty directory is on an
712
+ * inode that is no longer the run's, which its post-launch look can see only while the path is not the run's again;
713
+ * that residual is stated in `INT-SANDBOX-CONTRACT`. Anything else at the name (a file, a non-empty directory, a
714
+ * link) is never removed, and the tombstone is then held (`clearTombstone` never deletes a pinned one).
715
+ */
716
+ function restoreTombstone(tomb, dir) {
717
+ try {
718
+ fs.lstatSync(dir);
719
+ try {
720
+ fs.rmdirSync(dir);
721
+ } catch {
722
+ return false;
723
+ }
724
+ } catch (err) {
725
+ if (err?.code !== "ENOENT") return false;
726
+ }
727
+ try {
728
+ fs.renameSync(tomb, dir);
729
+ return true;
730
+ } catch {
731
+ return false;
732
+ }
733
+ }
247
734
  }
248
735
 
249
736
  /**
@@ -254,16 +741,96 @@ export function makeSandboxReaper({
254
741
  * the base window while it lasts, and an unparseable `keepUntil` is treated as no pin rather than as
255
742
  * forever -- the direction that stays bounded.
256
743
  */
257
- function expiry(dir, fs, at, cutoff) {
258
- let manifest;
744
+ function expiry(read, at, retentionHours) {
745
+ // The pass's one read (`readRetained`). ENOENT, a manifest that does not parse, and one that cannot be read for
746
+ // good are all "no manifest", as they always were: nothing can open such a run (`resolveSandbox` refuses it). The
747
+ // runtime watch has already asked every runtime present about the last two, so a shell open on one is held above.
748
+ if (!Object.hasOwn(read ?? {}, "manifest")) return { expired: true, reason: "no-manifest" };
749
+ const deadline = sandboxDeadline(read.manifest, retentionHours);
750
+ if (deadline.until === null) return { expired: true, reason: "no-created-at" };
751
+ // `keepUntil` and `retainUntil` are instants the run ends AT; the window is the one it always was, a run created
752
+ // exactly `retentionHours` ago is still inside it.
753
+ if (deadline.source === "window") return deadline.until < at ? { expired: true, reason: "window" } : { expired: false };
754
+ return deadline.until <= at ? { expired: true, reason: deadline.source === "pin" ? "pin-expired" : "window" } : { expired: false };
755
+ }
756
+
757
+ /**
758
+ * When one retained run's window closes: `{ until, source }`, `until` in epoch ms or null (issue #446).
759
+ *
760
+ * ONE RULE, for the sweep and for every opener: the pin (`keepUntil`) while one is set, whatever else the manifest
761
+ * says; otherwise the EARLIER of the deadline the worker wrote when it retained the run (`retainUntil`) and
762
+ * `createdAt` plus `retentionHours`, the READER's current window. An unparseable `keepUntil` is no pin rather than
763
+ * forever, the direction that stays bounded, and a manifest with no usable `createdAt` has no deadline at all
764
+ * (`until: null`), which the sweep expires and an opener refuses.
765
+ *
766
+ * WHY THE EARLIER OF THE TWO, both halves (issue #446, decided). The window's half: SHORTENING APPLIES. An operator who
767
+ * lowers PI_SANDBOX_RETENTION_HOURS, or sets it to 0, is cleaning up, and that has always swept what an earlier
768
+ * setting kept. `retainUntil`'s half: a reader whose window is LONGER than the worker's (the panel, or a shell with its
769
+ * own setting) must not call open a run the worker is about to delete, which is window 1 of #446 across two
770
+ * environments. The one case neither half covers is a worker whose CURRENT window is shorter than both the opener's
771
+ * and the run's `retainUntil` (the worker's window lowered after retention, the opener's not): the opener then admits
772
+ * a run the next pass deletes. That residual is accepted and caught after the fact, by the opener's post-launch look
773
+ * (`openSandbox`), which removes the container and reports `swept-at-launch` (`INT-SANDBOX-CONTRACT`).
774
+ */
775
+ export function sandboxDeadline(manifest, retentionHours) {
776
+ const keepUntil = Date.parse(manifest?.keepUntil ?? "");
777
+ if (Number.isFinite(keepUntil)) return { until: keepUntil, source: "pin" };
778
+ // A string or nothing: `Date.parse(5)` reads a number as a YEAR (gate round 1), and this key is host-written.
779
+ const retainUntil = typeof manifest?.retainUntil === "string" ? Date.parse(manifest.retainUntil) : NaN;
780
+ const createdAt = Date.parse(manifest?.createdAt ?? "");
781
+ if (!Number.isFinite(createdAt)) return { until: null, source: "no-created-at" };
782
+ const windowUntil = createdAt + (Number(retentionHours) || 0) * HOUR_MS;
783
+ // Strictly earlier, so the ordinary case (the same window both times, the two equal) keeps the window's own edge.
784
+ if (Number.isFinite(retainUntil) && retainUntil < windowUntil) return { until: retainUntil, source: "retain" };
785
+ return { until: windowUntil, source: "window" };
786
+ }
787
+
788
+ /**
789
+ * The sweep's own verdict for a manifest in `readManifest`'s shape (issue #446): `{ expired, reason? }` at `at`. The
790
+ * adapter over the pass's `readRetained` shape, so an opener or the panel asks the question the sweep asks and never a
791
+ * copy of it. A null manifest is `no-manifest`, as the sweep reads one.
792
+ */
793
+ export function sandboxExpiry(manifest, { at, retentionHours }) {
794
+ return expiry(manifest ? { manifest } : {}, at, retentionHours);
795
+ }
796
+
797
+ /**
798
+ * One retained entry's manifest, read once for the whole pass: `{ manifest }`, `{ absent: true }` (no manifest file, or
799
+ * not a directory at all), `{ transient: true, code }` (a read that failed for a moment, `TRANSIENT_READ_ERRORS`),
800
+ * `{ unparsed: true }` (the file does not parse) or `{ unreadable: true, code }` (any other failure, EACCES say).
801
+ * `lstat` FIRST, as the pass itself does: a symlink planted here must not be followed to read a host file.
802
+ */
803
+ export function readRetained(fs, dir) {
804
+ try {
805
+ if (!fs.lstatSync(dir).isDirectory()) return { absent: true };
806
+ } catch (err) {
807
+ return TRANSIENT_READ_ERRORS.has(err?.code) ? { transient: true, code: err.code } : { absent: true };
808
+ }
809
+ let text;
810
+ try {
811
+ text = String(fs.readFileSync(join(dir, SANDBOX_MANIFEST), "utf8"));
812
+ } catch (err) {
813
+ if (err?.code === "ENOENT" || err?.code === "ENOTDIR") return { absent: true };
814
+ if (TRANSIENT_READ_ERRORS.has(err?.code)) return { transient: true, code: err.code };
815
+ return { unreadable: true, code: err?.code ?? null };
816
+ }
817
+ // `text` rides along so the delete can compare the pass's read with a fresh one byte for byte (`sameRead`).
259
818
  try {
260
- manifest = JSON.parse(fs.readFileSync(join(dir, SANDBOX_MANIFEST), "utf8"));
819
+ return { manifest: JSON.parse(text), text };
261
820
  } catch {
262
- return { expired: true, reason: "no-manifest" };
821
+ return { unparsed: true, text };
263
822
  }
264
- const keepUntil = Date.parse(manifest?.keepUntil ?? "");
265
- if (Number.isFinite(keepUntil)) return keepUntil <= at ? { expired: true, reason: "pin-expired" } : { expired: false };
266
- const createdAt = Date.parse(manifest?.createdAt ?? "");
267
- if (!Number.isFinite(createdAt)) return { expired: true, reason: "no-created-at" };
268
- return createdAt < cutoff ? { expired: true, reason: "window" } : { expired: false };
823
+ }
824
+
825
+ /**
826
+ * Whether two `readRetained` results are the same read: the same kind, and for a manifest (parsed or not) the same
827
+ * bytes. Two absences are the same, and so are two failures with the same errno; a transient failure never reaches here.
828
+ */
829
+ function sameRead(a, b) {
830
+ if (a?.absent || b?.absent) return a?.absent === true && b?.absent === true;
831
+ // An unreadable file matches only the same failure (EACCES then EACCES): nothing changed between the reads, and a
832
+ // file that stays unreadable must still be swept once every runtime has said no sandbox is open on it.
833
+ if (a?.unreadable || b?.unreadable) return a?.unreadable === true && b?.unreadable === true && a.code === b.code;
834
+ if (typeof a?.text !== "string" || typeof b?.text !== "string") return false;
835
+ return a.text === b.text && Object.hasOwn(a, "manifest") === Object.hasOwn(b, "manifest");
269
836
  }