@dogfood-lab/ingest 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/load-context.js CHANGED
@@ -201,6 +201,25 @@ export function localScenarioFetcher(repoRoot) {
201
201
  */
202
202
  export const GITHUB_SCENARIO_FETCH_TIMEOUT_MS = 30000;
203
203
 
204
+ /**
205
+ * Default number of fetch attempts (INGEST-PROACT-003). A transient GitHub-API
206
+ * fault (5xx, 429, network blip) should not turn a loadable scenario into a hard
207
+ * rejection; a small bounded retry rides out the hiccup. Kept small — three
208
+ * attempts is enough to clear a momentary throttle without amplifying a real
209
+ * outage into a multi-minute stall (the AbortController timeout still bounds each
210
+ * attempt, and a 404/invalid-id is never retried).
211
+ */
212
+ export const GITHUB_SCENARIO_FETCH_ATTEMPTS = 3;
213
+
214
+ /** Initial backoff before the first retry (ms); doubles per attempt, capped. */
215
+ const RETRY_BASE_MS = 250;
216
+ const RETRY_MAX_MS = 2000;
217
+
218
+ /** Default async backoff. Injectable (`opts.sleepImpl`) so tests run instantly. */
219
+ function defaultSleep(ms) {
220
+ return new Promise((r) => setTimeout(r, ms));
221
+ }
222
+
204
223
  /**
205
224
  * GitHub scenario fetcher. Loads scenario definitions from a source repo
206
225
  * via the GitHub API at a specific commit SHA.
@@ -219,28 +238,45 @@ export const GITHUB_SCENARIO_FETCH_TIMEOUT_MS = 30000;
219
238
  * Both surfaces honour the per-request AbortController timeout
220
239
  * (`GITHUB_SCENARIO_FETCH_TIMEOUT_MS`, overridable via `opts.timeoutMs`).
221
240
  *
241
+ * INGEST-PROACT-003: each call makes up to `opts.attempts`
242
+ * (`GITHUB_SCENARIO_FETCH_ATTEMPTS`) tries with exponential backoff, retrying
243
+ * ONLY the transient classes — request timeout, HTTP 5xx, HTTP 429, and network
244
+ * rejects. A 404 (`not_found`), an `invalid_id`, and a `parse_error` are
245
+ * DEFINITIVE answers and are returned immediately without a retry (mirrors the
246
+ * EPERM/EBUSY-only discipline in `lib/rename-with-retry.js`). The backoff sleep
247
+ * is injectable (`opts.sleepImpl`) so tests do not actually wait.
248
+ *
222
249
  * @param {string} token - GitHub PAT
223
250
  * @param {string} repoSlug - e.g. "mcp-tool-shop-org/shipcheck"
224
251
  * @param {string} commitSha - Commit to fetch scenarios from
225
- * @param {{ timeoutMs?: number, fetchImpl?: typeof fetch }} [opts]
252
+ * @param {{ timeoutMs?: number, fetchImpl?: typeof fetch, attempts?: number, sleepImpl?: (ms: number) => Promise<void> }} [opts]
226
253
  * @returns {{ fetch(scenarioId: string): Promise<object|null>, fetchWithReason(scenarioId: string): Promise<{ scenario: object|null, reason?: string }> }}
227
254
  */
228
255
  export function githubScenarioFetcher(token, repoSlug, commitSha, opts = {}) {
229
256
  const timeoutMs = opts.timeoutMs ?? GITHUB_SCENARIO_FETCH_TIMEOUT_MS;
230
257
  const fetchImpl = opts.fetchImpl ?? ((url, init) => globalThis.fetch(url, init));
258
+ const attempts = opts.attempts ?? GITHUB_SCENARIO_FETCH_ATTEMPTS;
259
+ const sleep = opts.sleepImpl ?? defaultSleep;
231
260
 
232
261
  const [org, repo] = repoSlug.split('/');
233
- if (!org || !repo || isUnsafeSegment(org) || isUnsafeSegment(repo)) {
262
+ // commitSha is interpolated into the authenticated (Bearer-token) GitHub API
263
+ // URL's `?ref=` — a shape guard (lowercase-hex, 7–40 chars) refuses anything
264
+ // that could re-target the ref or inject query params, matching how org/repo
265
+ // and scenarioId are guarded below. encodeURIComponent alone would NOT reject
266
+ // a re-targeted ref (e.g. a branch name), so the shape guard is the floor.
267
+ if (
268
+ !org || !repo || isUnsafeSegment(org) || isUnsafeSegment(repo) ||
269
+ typeof commitSha !== 'string' || !/^[0-9a-f]{7,40}$/.test(commitSha)
270
+ ) {
234
271
  return {
235
272
  async fetch() { return null; },
236
273
  async fetchWithReason() { return { scenario: null, reason: 'invalid_id' }; }
237
274
  };
238
275
  }
239
276
 
240
- async function fetchWithReason(scenarioId) {
241
- if (!/^[\w-]+$/.test(scenarioId)) {
242
- return { scenario: null, reason: 'invalid_id' };
243
- }
277
+ // One bounded attempt. Returns `{ scenario, reason, retryable }`; the loop
278
+ // below decides whether to retry on `retryable`.
279
+ async function attemptOnce(scenarioId) {
244
280
  const path = `dogfood/scenarios/${scenarioId}.yaml`;
245
281
  const url = `https://api.github.com/repos/${repoSlug}/contents/${path}?ref=${commitSha}`;
246
282
 
@@ -262,16 +298,20 @@ export function githubScenarioFetcher(token, repoSlug, commitSha, opts = {}) {
262
298
  signal: controller.signal
263
299
  });
264
300
  if (!resp.ok) {
265
- return { scenario: null, reason: 'not_found' };
301
+ // A 5xx server error or a 429 rate-limit is transient — retry. Any
302
+ // other non-ok (notably 404) is a definitive answer; do not retry.
303
+ const retryable = resp.status >= 500 || resp.status === 429;
304
+ return { scenario: null, reason: 'not_found', retryable };
266
305
  }
267
306
  text = await resp.text();
268
307
  } catch (err) {
269
308
  if (err && (err.name === 'AbortError' || err.code === 'ABORT_ERR')) {
270
- return { scenario: null, reason: 'timeout' };
309
+ // A timed-out request may succeed on a retry.
310
+ return { scenario: null, reason: 'timeout', retryable: true };
271
311
  }
272
- // Network reject, DNS failure, etc. — surface as not_found for
273
- // back-compat with the legacy null contract.
274
- return { scenario: null, reason: 'not_found' };
312
+ // Network reject, DNS failure, etc. — transient; surface as not_found
313
+ // for back-compat with the legacy null contract, but allow a retry.
314
+ return { scenario: null, reason: 'not_found', retryable: true };
275
315
  } finally {
276
316
  clearTimeout(timer);
277
317
  }
@@ -279,12 +319,28 @@ export function githubScenarioFetcher(token, repoSlug, commitSha, opts = {}) {
279
319
  try {
280
320
  const scenario = yaml.load(text);
281
321
  if (!scenario || typeof scenario !== 'object') {
282
- return { scenario: null, reason: 'parse_error' };
322
+ return { scenario: null, reason: 'parse_error', retryable: false };
283
323
  }
284
324
  return { scenario };
285
325
  } catch {
286
- return { scenario: null, reason: 'parse_error' };
326
+ return { scenario: null, reason: 'parse_error', retryable: false };
327
+ }
328
+ }
329
+
330
+ async function fetchWithReason(scenarioId) {
331
+ if (!/^[\w-]+$/.test(scenarioId)) {
332
+ return { scenario: null, reason: 'invalid_id' };
333
+ }
334
+
335
+ let last;
336
+ for (let i = 0; i < attempts; i++) {
337
+ last = await attemptOnce(scenarioId);
338
+ if (last.scenario || !last.retryable || i === attempts - 1) break;
339
+ await sleep(Math.min(RETRY_BASE_MS * (1 << i), RETRY_MAX_MS));
287
340
  }
341
+ // Strip the internal `retryable` flag from the public contract.
342
+ const { retryable: _drop, ...result } = last;
343
+ return result;
288
344
  }
289
345
 
290
346
  return {
package/package.json CHANGED
@@ -1,13 +1,15 @@
1
1
  {
2
2
  "name": "@dogfood-lab/ingest",
3
- "version": "1.4.0",
3
+ "version": "1.6.0",
4
4
  "type": "module",
5
5
  "description": "Ingestion pipeline for testing-os. Thin glue: dispatch → verifier → persist → indexes.",
6
6
  "main": "run.js",
7
7
  "exports": {
8
8
  ".": "./run.js",
9
9
  "./lib/*": "./lib/*",
10
- "./validate-record.js": "./validate-record.js"
10
+ "./anchor/*": "./anchor/*",
11
+ "./validate-record.js": "./validate-record.js",
12
+ "./verify-chain.js": "./verify-chain.js"
11
13
  },
12
14
  "scripts": {
13
15
  "test": "node --test",
@@ -18,8 +20,10 @@
18
20
  "persist.js",
19
21
  "rebuild-indexes.js",
20
22
  "validate-record.js",
23
+ "verify-chain.js",
21
24
  "load-context.js",
22
25
  "lib/",
26
+ "anchor/",
23
27
  "README.md",
24
28
  "LICENSE"
25
29
  ],
package/persist.js CHANGED
@@ -7,11 +7,13 @@
7
7
  */
8
8
 
9
9
  import { existsSync, mkdirSync, writeFileSync, renameSync, openSync, closeSync, unlinkSync } from 'node:fs';
10
- import { join, dirname } from 'node:path';
10
+ import { join, dirname, relative, sep } from 'node:path';
11
11
  import { randomBytes } from 'node:crypto';
12
12
 
13
13
  import { validateRecord } from './validate-record.js';
14
14
  import { isUnsafeSegment } from './lib/unsafe-segment.js';
15
+ import { submissionDigest } from './lib/integrity.js';
16
+ import { readChainHead, appendChainEntry } from './lib/chain-manifest.js';
15
17
 
16
18
  /**
17
19
  * Error thrown when writeRecord loses a TOCTOU race for the same canonical path.
@@ -125,6 +127,26 @@ export function writeRecord(record, repoRoot) {
125
127
  return { path, written: false };
126
128
  }
127
129
 
130
+ // Integrity chain v1 — stamp the tamper-evident integrity block BEFORE
131
+ // validating + writing, so the persisted record self-certifies.
132
+ //
133
+ // Serialized-ingest assumption (LOCKED CONTRACT step 4): ingest.yml is
134
+ // concurrency-serialized and writeRecord is synchronous, so reading the chain
135
+ // head and then appending after the write is race-free. A fork for
136
+ // truly-concurrent ingest (two writers assigning the same seq) is OUT OF SCOPE
137
+ // — see lib/chain-manifest.js. The order is: read head → compute digest over
138
+ // the record WITHOUT integrity (stable regardless of the block) → stamp
139
+ // integrity → write the record → append the manifest line. The append happens
140
+ // ONLY after the record write succeeds, and only on this real-write path
141
+ // (never on the duplicate short-circuit above).
142
+ const head = readChainHead(repoRoot);
143
+ const digest = submissionDigest(record);
144
+ record.integrity = {
145
+ submission_digest: digest,
146
+ prev_digest: head.submission_digest,
147
+ seq: head.seq + 1,
148
+ };
149
+
128
150
  // Enforce dogfood-record.schema.json BEFORE touching the filesystem.
129
151
  // Better to throw loudly than silently persist a malformed record — the
130
152
  // schema is the contract every downstream consumer relies on.
@@ -168,5 +190,23 @@ export function writeRecord(record, repoRoot) {
168
190
  throw err;
169
191
  }
170
192
 
193
+ // Append the chain ledger line AFTER the record write succeeds. The manifest
194
+ // append is atomic (temp+rename rewrite — see lib/chain-manifest.js); a torn
195
+ // append cannot leave a half-line. `path` field is the record path RELATIVE to
196
+ // repoRoot, forward-slashed, so the ledger is portable across OSes and a line
197
+ // copy-pasted into a raw.githubusercontent URL is not a broken link (mirrors
198
+ // the posixify-at-the-boundary doctrine in run.js / rebuild-indexes.js).
199
+ const relPath = relative(repoRoot, path).split(sep).join('/');
200
+ appendChainEntry(repoRoot, {
201
+ seq: record.integrity.seq,
202
+ run_id: record.run_id,
203
+ repo: record.repo,
204
+ status: record.verification?.status ?? 'accepted',
205
+ path: relPath,
206
+ submission_digest: record.integrity.submission_digest,
207
+ prev_digest: record.integrity.prev_digest,
208
+ persisted_at: new Date().toISOString(),
209
+ });
210
+
171
211
  return { path, written: true };
172
212
  }
@@ -8,7 +8,8 @@
8
8
  *
9
9
  * Regenerated on every accepted/rejected write in Phase 1.
10
10
  *
11
- * Multi-file commit-group atomicity (W3-PIPE-002):
11
+ * Multi-file commit-group: crash/IO-failure RECOVERY-atomic, NOT reader-atomic
12
+ * (W3-PIPE-002):
12
13
  * The 3 indexes are written together via a two-phase commit pattern. Phase 1
13
14
  * stages all 3 files to temp paths AND records them in a journal file. Phase 2
14
15
  * renames each temp into its final location, then deletes the journal. If the
@@ -17,10 +18,27 @@
17
18
  * is idempotent (it scans records/ end-to-end), so re-running is the correct
18
19
  * recovery action.
19
20
  *
20
- * Pattern reference: choke-point fix (Pattern #4) for multi-file atomicity.
21
+ * IMPORTANT — what "atomicity" means here. The guarantee is RECOVERY-atomic,
22
+ * not READER-atomic. Phase 2 promotes the temps with a per-leg `renameSync`
23
+ * (each rename is individually atomic), but the GROUP is not promoted under a
24
+ * single atomic operation. During the promote window — and during the heal
25
+ * window after a mid-promote IO failure (ENOSPC/EACCES after the first
26
+ * final is renamed but a later one is not) — a concurrent reader CAN observe
27
+ * the index group in a mutually-inconsistent intermediate state (e.g. an
28
+ * already-promoted latest-by-repo.json against a not-yet-promoted failing.json).
29
+ * The catch on a promote failure does NOT roll back already-promoted finals;
30
+ * it preserves the journal and emits a structured error event so an operator
31
+ * can force an immediate rebuild before the next scheduled run heals it. The
32
+ * design is sound because the only writer (the ingest pipeline) serializes
33
+ * rebuilds and `rebuildIndexes` is synchronous — there is no in-flight reader
34
+ * that races a writer mid-promote within a single process. If you ever need
35
+ * true reader-atomicity (a reader that NEVER sees a torn group), this design
36
+ * must change (e.g. swap a single directory symlink, or version the index dir).
37
+ *
38
+ * Pattern reference: choke-point fix (Pattern #4) for multi-file recovery.
21
39
  * Single-file `atomicWriteFileSync` (lib/atomic-write.js) handles each leg;
22
- * the journal handles the cross-file boundary. The single-file helper is
23
- * the same one Class #6 helper-adoption-sweep enforces as canonical for
40
+ * the journal handles the cross-file recovery boundary. The single-file helper
41
+ * is the same one Class #6 helper-adoption-sweep enforces as canonical for
24
42
  * temp+rename writes under `packages/ingest/`.
25
43
  */
26
44
 
@@ -98,6 +116,72 @@ export function findJsonFiles(dir) {
98
116
  return results;
99
117
  }
100
118
 
119
+ /**
120
+ * Probe whether `dir` exists but is unreadable at its OWN level (EACCES /
121
+ * Windows lock / ENOTDIR) — as distinct from a deep leaf failing mid-walk.
122
+ *
123
+ * ingest-B-002: `findJsonFiles` deliberately degrades an unreadable subtree to
124
+ * "those records are missing" and returns `[]`. That is correct for a single
125
+ * locked LEAF, but catastrophic for the records/ ROOT: a transiently-locked
126
+ * root makes the WHOLE corpus invisible, and an unguarded rebuild would then
127
+ * overwrite every index with empty content. This probe lets `rebuildIndexes`
128
+ * tell the two apart so it can REFUSE to clobber good indexes when the root
129
+ * itself is the thing that failed. A non-existent dir is NOT unreadable — that
130
+ * is the legitimate empty-corpus case, which must still rebuild empty indexes.
131
+ *
132
+ * @param {string} dir
133
+ * @returns {{ unreadable: boolean, code: string|null, error: string|null }}
134
+ */
135
+ function probeDirReadable(dir) {
136
+ if (!existsSync(dir)) return { unreadable: false, code: null, error: null };
137
+ try {
138
+ readdirSync(dir);
139
+ return { unreadable: false, code: null, error: null };
140
+ } catch (err) {
141
+ return {
142
+ unreadable: true,
143
+ code: err && err.code ? err.code : 'readdir_failed',
144
+ error: err && err.message ? err.message : String(err),
145
+ };
146
+ }
147
+ }
148
+
149
+ /**
150
+ * Read the prior committed latest-by-repo.json so a rebuild can tell whether
151
+ * the index it is about to overwrite currently has content. Used by the
152
+ * ingest-B-002 refuse-to-overwrite guard: an empty scan is only suspicious if
153
+ * the prior index was non-empty. A missing or unparseable prior index counts
154
+ * as "no prior content" (the legitimate first-run / empty-corpus case).
155
+ *
156
+ * @param {string} latestPath
157
+ * @returns {boolean} true if the prior index existed and held at least one repo
158
+ */
159
+ function priorIndexHasContent(latestPath) {
160
+ if (!existsSync(latestPath)) return false;
161
+ try {
162
+ const prior = JSON.parse(readFileSync(latestPath, 'utf-8'));
163
+ return prior && typeof prior === 'object' && Object.keys(prior).length > 0;
164
+ } catch {
165
+ // Unparseable prior index — treat as no usable content so a corrupt index
166
+ // never wedges the rebuild into a permanent refuse state.
167
+ return false;
168
+ }
169
+ }
170
+
171
+ /**
172
+ * Whether the freshly-built latest-by-repo map has no repos. Used by the
173
+ * ingest-B-002 refuse-to-overwrite guard to recognise an empty scan. A scan
174
+ * can be empty because there are genuinely no accepted records (legitimate)
175
+ * or because the corpus was invisible (a transiently-locked records tree) —
176
+ * the guard combines this with `priorIndexHasContent` to tell them apart.
177
+ *
178
+ * @param {object} latestByRepo
179
+ * @returns {boolean}
180
+ */
181
+ function latestByRepoIsEmpty(latestByRepo) {
182
+ return !latestByRepo || Object.keys(latestByRepo).length === 0;
183
+ }
184
+
101
185
  /**
102
186
  * Load and parse a record file.
103
187
  *
@@ -131,8 +215,20 @@ export function rebuildIndexes(repoRoot, options = {}) {
131
215
  const indexDir = join(repoRoot, 'indexes');
132
216
  mkdirSync(indexDir, { recursive: true });
133
217
 
134
- // Collect all records (accepted + rejected)
218
+ // ingest-B-002: capture two facts BEFORE the scan so we can refuse to clobber
219
+ // good indexes with empty ones when the corpus is invisible rather than empty.
220
+ // 1. Is the records/ ROOT itself unreadable (vs a deep leaf, vs absent)?
221
+ // A locked root makes the WHOLE corpus invisible — findJsonFiles would
222
+ // return [] after one low `dir_unreadable` warn, and an unguarded
223
+ // commit-group would then overwrite every index with {}.
224
+ // 2. Did the PRIOR latest-by-repo.json have content? An empty scan is only
225
+ // suspicious if there was something to lose; a legitimately empty
226
+ // first-run corpus must still write empty indexes.
135
227
  const recordsDir = join(repoRoot, 'records');
228
+ const latestPath = join(indexDir, 'latest-by-repo.json');
229
+ const rootProbe = probeDirReadable(recordsDir);
230
+ const hadPriorIndex = priorIndexHasContent(latestPath);
231
+
136
232
  const acceptedFiles = findJsonFiles(recordsDir)
137
233
  .filter(f => {
138
234
  const rel = relative(recordsDir, f);
@@ -268,9 +364,49 @@ export function rebuildIndexes(repoRoot, options = {}) {
268
364
  }
269
365
  }
270
366
 
367
+ // ingest-B-002: REFUSE to overwrite good indexes with empty ones when the
368
+ // corpus was invisible rather than genuinely empty. Two refuse conditions:
369
+ // - records_root_unreadable: the records/ ROOT itself failed to read
370
+ // (EACCES / Windows lock / ENOTDIR). The entire corpus is invisible —
371
+ // committing now would wipe every index. This is distinct from a single
372
+ // locked leaf, which findJsonFiles already degrades to a partial scan.
373
+ // - empty_scan_with_prior_index: the root read fine but the scan found
374
+ // zero accepted records while the prior latest-by-repo had content. The
375
+ // records likely vanished transiently; clobbering loses the portfolio.
376
+ // A legitimately empty corpus (no accepted records AND no prior content) is
377
+ // NOT refused — it must still write empty indexes (first-run case). We skip
378
+ // the commit-group and emit a structured, greppable event so the operator
379
+ // sees the refusal loudly instead of a silently-emptied portfolio.
380
+ const noAcceptedScanned = latestByRepoIsEmpty(latestByRepo);
381
+ if (rootProbe.unreadable || (noAcceptedScanned && hadPriorIndex)) {
382
+ const reason = rootProbe.unreadable
383
+ ? 'records_root_unreadable'
384
+ : 'empty_scan_with_prior_index';
385
+ // A root IO failure is an operator-actionable error (the corpus is gone);
386
+ // an empty scan over a readable root is a warn (recoverable next run).
387
+ logStage(rootProbe.unreadable ? 'error' : 'warn', {
388
+ kind: 'index_rebuild_skipped',
389
+ reason,
390
+ records_dir: recordsDir,
391
+ accepted_scanned: acceptedFiles.length,
392
+ prior_index_non_empty: hadPriorIndex,
393
+ root_error_code: rootProbe.code,
394
+ error: rootProbe.error,
395
+ });
396
+ return {
397
+ latestByRepo,
398
+ failing,
399
+ stale,
400
+ accepted: acceptedFiles.length,
401
+ rejected: rejectedFiles.length,
402
+ corrupted,
403
+ skipped,
404
+ skippedCommit: reason,
405
+ };
406
+ }
407
+
271
408
  // Write indexes via commit-group two-phase commit. See module header
272
409
  // for the full design rationale.
273
- const latestPath = join(indexDir, 'latest-by-repo.json');
274
410
  const failingPath = join(indexDir, 'failing.json');
275
411
  const stalePath = join(indexDir, 'stale.json');
276
412
 
@@ -307,15 +443,23 @@ export function rebuildIndexes(repoRoot, options = {}) {
307
443
  * AND records them in a journal first; then renames them in caller-given
308
444
  * order. The journal is deleted only after every rename succeeds.
309
445
  *
310
- * Crash semantics:
311
- * - Crash during STAGE phase: every staged temp is unlinked in the catch
446
+ * Crash / IO-failure semantics (RECOVERY-atomic, not reader-atomic):
447
+ * - Failure during STAGE phase: every staged temp is unlinked in the catch
312
448
  * block; the journal (if written) is unlinked too. No partial visible
313
- * state.
314
- * - Crash during PROMOTE phase: any successfully-renamed file is at its
315
- * final path; remaining temps are still next to their finals. The
316
- * journal still exists. Next run's `cleanupCrashedJournals` deletes
317
- * residual temps and the journal; the next normal `rebuildIndexes`
318
- * call rewrites all 3 indexes from scratch (idempotent).
449
+ * state — no final was touched.
450
+ * - Failure during PROMOTE phase: any successfully-renamed file is at its
451
+ * final path with its NEW content; remaining temps are still next to
452
+ * their (still-OLD) finals. The group is therefore mutually inconsistent
453
+ * until healed — a reader in this window sees a torn group. We do NOT
454
+ * roll back the already-promoted finals (their prior content was already
455
+ * overwritten by the atomic rename — there is nothing to roll back to
456
+ * without re-reading the journal). Instead we emit a structured
457
+ * `logStage('error', { kind: 'commit_group_partial_promote', ... })`
458
+ * naming which finals were promoted vs left stale so an operator can
459
+ * force an immediate rebuild, and we preserve the journal. Next run's
460
+ * `cleanupCrashedJournals` deletes residual temps and the journal; the
461
+ * next normal `rebuildIndexes` call rewrites all 3 indexes from scratch
462
+ * (idempotent), which is what heals the torn group.
319
463
  *
320
464
  * Why journal-then-rename rather than journal-only: the rename phase needs
321
465
  * to be the visible commit point. A journal-only design would require
@@ -339,8 +483,12 @@ function commitGroupRename(indexDir, entries) {
339
483
 
340
484
  // Write journal AFTER staging so it never points at a non-existent temp.
341
485
  // Atomic write of the journal itself: writeFileSync directly is fine here
342
- // because the journal is process-private (the pid suffix guarantees no
343
- // collision with concurrent rebuilds).
486
+ // because the journal is process-private — the pid + random suffix make
487
+ // the filename collision-free, and `cleanupCrashedJournals` is pid-aware
488
+ // (it skips journals whose pid is a still-live process), so a future
489
+ // concurrent rebuild's in-flight journal is never reaped out from under
490
+ // it. The temp `entries` it lists are equally collision-free (each carries
491
+ // its own random suffix from `stageWriteFileSync`).
344
492
  writeFileSync(
345
493
  journalPath,
346
494
  JSON.stringify({
@@ -374,6 +522,24 @@ function commitGroupRename(indexDir, entries) {
374
522
  // (their previous content is already overwritten — the rename was
375
523
  // atomic at each individual leg, just not as a group). The next run
376
524
  // is idempotent and will rewrite all three from scratch.
525
+ //
526
+ // ingest-A-001: the group is now reader-inconsistent (promoted finals
527
+ // carry new content; stale finals carry old content). Name which finals
528
+ // are which in a structured error event so an operator can force an
529
+ // immediate rebuild rather than wait for the next scheduled run to heal
530
+ // the torn group.
531
+ const promoted = stagedTmps.slice(0, promotedCount).map((e) => e.finalPath);
532
+ const stale = stagedTmps.slice(promotedCount).map((e) => e.finalPath);
533
+ logStage('error', {
534
+ kind: 'commit_group_partial_promote',
535
+ reason: err && err.code ? err.code : 'promote_failed',
536
+ promoted_count: promotedCount,
537
+ total: stagedTmps.length,
538
+ promoted_finals: promoted,
539
+ stale_finals: stale,
540
+ journal: journalPath,
541
+ error: err && err.message ? err.message : String(err),
542
+ });
377
543
  throw new Error(
378
544
  `commitGroupRename: promote failed after ${promotedCount}/${stagedTmps.length} files; ` +
379
545
  `journal preserved at ${journalPath} for next-run cleanup. Original error: ${err.message}`
@@ -387,11 +553,42 @@ function commitGroupRename(indexDir, entries) {
387
553
  try { unlinkSync(journalPath); } catch { /* will be cleaned next run */ }
388
554
  }
389
555
 
556
+ /**
557
+ * Probe whether a pid is still a live process. `process.kill(pid, 0)` sends
558
+ * no signal — it only performs the permission/existence check, throwing
559
+ * ESRCH when the pid is dead. An EPERM means the process exists but is owned
560
+ * by another user; that still counts as "live" for our purpose (do not reap
561
+ * its journal). Any other error (or a non-integer pid) is treated as "not
562
+ * provably live" so a malformed journal never blocks its own cleanup.
563
+ *
564
+ * @param {unknown} pid
565
+ * @returns {boolean}
566
+ */
567
+ function isProcessAlive(pid) {
568
+ if (!Number.isInteger(pid) || pid <= 0) return false;
569
+ try {
570
+ process.kill(pid, 0);
571
+ return true;
572
+ } catch (err) {
573
+ return err && err.code === 'EPERM';
574
+ }
575
+ }
576
+
390
577
  /**
391
578
  * Find and clean up any in-progress journals from previous runs. Each journal
392
579
  * lists the temp paths that were staged; we unlink any that still exist
393
580
  * (they are residue from a crashed run) and delete the journal.
394
581
  *
582
+ * ingest-A-002: cleanup is PID-AWARE. A journal whose `pid` is a still-live
583
+ * process is the in-flight recovery state of a concurrent rebuild — reaping
584
+ * it would delete that run's temps and journal mid-flight. Today the only
585
+ * writer serializes rebuilds and `rebuildIndexes` is synchronous, so no live
586
+ * sibling journal exists at Phase-0 cleanup time; this guard makes the design
587
+ * correct (not merely safe-by-serialization) so a future maintainer who adds
588
+ * concurrency does not silently corrupt a peer. A dead pid, a missing/
589
+ * malformed pid, or an unreadable journal is still reaped — that is the
590
+ * crashed-run residue this function exists to clear.
591
+ *
395
592
  * Idempotent: on a clean filesystem it's a no-op; on a crashed-mid-promote
396
593
  * filesystem it cleans the slate so the upcoming `commitGroupRename` can
397
594
  * stage fresh temps without colliding.
@@ -412,6 +609,11 @@ function cleanupCrashedJournals(indexDir) {
412
609
  // referenced will linger but they're harmless (they have a unique
413
610
  // suffix that won't be re-used).
414
611
  }
612
+ // Skip a journal owned by a still-live process — it belongs to a
613
+ // concurrent rebuild's in-flight recovery state, not crashed residue.
614
+ if (parsed && isProcessAlive(parsed.pid) && parsed.pid !== process.pid) {
615
+ continue;
616
+ }
415
617
  if (parsed && Array.isArray(parsed.entries)) {
416
618
  for (const e of parsed.entries) {
417
619
  if (e && typeof e.tmpPath === 'string') {