@dogfood-lab/ingest 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -22
- package/anchor/cli.js +215 -0
- package/anchor/compute-root.js +302 -0
- package/anchor/config.js +44 -0
- package/anchor/merkle.js +123 -0
- package/anchor/post-anchor.js +328 -0
- package/anchor/verify-anchor.js +358 -0
- package/lib/chain-manifest.js +106 -0
- package/lib/integrity.js +98 -0
- package/load-context.js +69 -13
- package/package.json +6 -2
- package/persist.js +41 -1
- package/rebuild-indexes.js +218 -16
- package/run.js +190 -11
- package/verify-chain.js +316 -0
package/load-context.js
CHANGED
|
@@ -201,6 +201,25 @@ export function localScenarioFetcher(repoRoot) {
|
|
|
201
201
|
*/
|
|
202
202
|
export const GITHUB_SCENARIO_FETCH_TIMEOUT_MS = 30000;
|
|
203
203
|
|
|
204
|
+
/**
|
|
205
|
+
* Default number of fetch attempts (INGEST-PROACT-003). A transient GitHub-API
|
|
206
|
+
* fault (5xx, 429, network blip) should not turn a loadable scenario into a hard
|
|
207
|
+
* rejection; a small bounded retry rides out the hiccup. Kept small — three
|
|
208
|
+
* attempts is enough to clear a momentary throttle without amplifying a real
|
|
209
|
+
* outage into a multi-minute stall (the AbortController timeout still bounds each
|
|
210
|
+
* attempt, and a 404/invalid-id is never retried).
|
|
211
|
+
*/
|
|
212
|
+
export const GITHUB_SCENARIO_FETCH_ATTEMPTS = 3;
|
|
213
|
+
|
|
214
|
+
/** Initial backoff before the first retry (ms); doubles per attempt, capped. */
|
|
215
|
+
const RETRY_BASE_MS = 250;
|
|
216
|
+
const RETRY_MAX_MS = 2000;
|
|
217
|
+
|
|
218
|
+
/** Default async backoff. Injectable (`opts.sleepImpl`) so tests run instantly. */
|
|
219
|
+
function defaultSleep(ms) {
|
|
220
|
+
return new Promise((r) => setTimeout(r, ms));
|
|
221
|
+
}
|
|
222
|
+
|
|
204
223
|
/**
|
|
205
224
|
* GitHub scenario fetcher. Loads scenario definitions from a source repo
|
|
206
225
|
* via the GitHub API at a specific commit SHA.
|
|
@@ -219,28 +238,45 @@ export const GITHUB_SCENARIO_FETCH_TIMEOUT_MS = 30000;
|
|
|
219
238
|
* Both surfaces honour the per-request AbortController timeout
|
|
220
239
|
* (`GITHUB_SCENARIO_FETCH_TIMEOUT_MS`, overridable via `opts.timeoutMs`).
|
|
221
240
|
*
|
|
241
|
+
* INGEST-PROACT-003: each call makes up to `opts.attempts`
|
|
242
|
+
* (`GITHUB_SCENARIO_FETCH_ATTEMPTS`) tries with exponential backoff, retrying
|
|
243
|
+
* ONLY the transient classes — request timeout, HTTP 5xx, HTTP 429, and network
|
|
244
|
+
* rejects. A 404 (`not_found`), an `invalid_id`, and a `parse_error` are
|
|
245
|
+
* DEFINITIVE answers and are returned immediately without a retry (mirrors the
|
|
246
|
+
* EPERM/EBUSY-only discipline in `lib/rename-with-retry.js`). The backoff sleep
|
|
247
|
+
* is injectable (`opts.sleepImpl`) so tests do not actually wait.
|
|
248
|
+
*
|
|
222
249
|
* @param {string} token - GitHub PAT
|
|
223
250
|
* @param {string} repoSlug - e.g. "mcp-tool-shop-org/shipcheck"
|
|
224
251
|
* @param {string} commitSha - Commit to fetch scenarios from
|
|
225
|
-
* @param {{ timeoutMs?: number, fetchImpl?: typeof fetch }} [opts]
|
|
252
|
+
* @param {{ timeoutMs?: number, fetchImpl?: typeof fetch, attempts?: number, sleepImpl?: (ms: number) => Promise<void> }} [opts]
|
|
226
253
|
* @returns {{ fetch(scenarioId: string): Promise<object|null>, fetchWithReason(scenarioId: string): Promise<{ scenario: object|null, reason?: string }> }}
|
|
227
254
|
*/
|
|
228
255
|
export function githubScenarioFetcher(token, repoSlug, commitSha, opts = {}) {
|
|
229
256
|
const timeoutMs = opts.timeoutMs ?? GITHUB_SCENARIO_FETCH_TIMEOUT_MS;
|
|
230
257
|
const fetchImpl = opts.fetchImpl ?? ((url, init) => globalThis.fetch(url, init));
|
|
258
|
+
const attempts = opts.attempts ?? GITHUB_SCENARIO_FETCH_ATTEMPTS;
|
|
259
|
+
const sleep = opts.sleepImpl ?? defaultSleep;
|
|
231
260
|
|
|
232
261
|
const [org, repo] = repoSlug.split('/');
|
|
233
|
-
|
|
262
|
+
// commitSha is interpolated into the authenticated (Bearer-token) GitHub API
|
|
263
|
+
// URL's `?ref=` — a shape guard (lowercase-hex, 7–40 chars) refuses anything
|
|
264
|
+
// that could re-target the ref or inject query params, matching how org/repo
|
|
265
|
+
// and scenarioId are guarded below. encodeURIComponent alone would NOT reject
|
|
266
|
+
// a re-targeted ref (e.g. a branch name), so the shape guard is the floor.
|
|
267
|
+
if (
|
|
268
|
+
!org || !repo || isUnsafeSegment(org) || isUnsafeSegment(repo) ||
|
|
269
|
+
typeof commitSha !== 'string' || !/^[0-9a-f]{7,40}$/.test(commitSha)
|
|
270
|
+
) {
|
|
234
271
|
return {
|
|
235
272
|
async fetch() { return null; },
|
|
236
273
|
async fetchWithReason() { return { scenario: null, reason: 'invalid_id' }; }
|
|
237
274
|
};
|
|
238
275
|
}
|
|
239
276
|
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
}
|
|
277
|
+
// One bounded attempt. Returns `{ scenario, reason, retryable }`; the loop
|
|
278
|
+
// below decides whether to retry on `retryable`.
|
|
279
|
+
async function attemptOnce(scenarioId) {
|
|
244
280
|
const path = `dogfood/scenarios/${scenarioId}.yaml`;
|
|
245
281
|
const url = `https://api.github.com/repos/${repoSlug}/contents/${path}?ref=${commitSha}`;
|
|
246
282
|
|
|
@@ -262,16 +298,20 @@ export function githubScenarioFetcher(token, repoSlug, commitSha, opts = {}) {
|
|
|
262
298
|
signal: controller.signal
|
|
263
299
|
});
|
|
264
300
|
if (!resp.ok) {
|
|
265
|
-
|
|
301
|
+
// A 5xx server error or a 429 rate-limit is transient — retry. Any
|
|
302
|
+
// other non-ok (notably 404) is a definitive answer; do not retry.
|
|
303
|
+
const retryable = resp.status >= 500 || resp.status === 429;
|
|
304
|
+
return { scenario: null, reason: 'not_found', retryable };
|
|
266
305
|
}
|
|
267
306
|
text = await resp.text();
|
|
268
307
|
} catch (err) {
|
|
269
308
|
if (err && (err.name === 'AbortError' || err.code === 'ABORT_ERR')) {
|
|
270
|
-
|
|
309
|
+
// A timed-out request may succeed on a retry.
|
|
310
|
+
return { scenario: null, reason: 'timeout', retryable: true };
|
|
271
311
|
}
|
|
272
|
-
// Network reject, DNS failure, etc. — surface as not_found
|
|
273
|
-
// back-compat with the legacy null contract.
|
|
274
|
-
return { scenario: null, reason: 'not_found' };
|
|
312
|
+
// Network reject, DNS failure, etc. — transient; surface as not_found
|
|
313
|
+
// for back-compat with the legacy null contract, but allow a retry.
|
|
314
|
+
return { scenario: null, reason: 'not_found', retryable: true };
|
|
275
315
|
} finally {
|
|
276
316
|
clearTimeout(timer);
|
|
277
317
|
}
|
|
@@ -279,12 +319,28 @@ export function githubScenarioFetcher(token, repoSlug, commitSha, opts = {}) {
|
|
|
279
319
|
try {
|
|
280
320
|
const scenario = yaml.load(text);
|
|
281
321
|
if (!scenario || typeof scenario !== 'object') {
|
|
282
|
-
return { scenario: null, reason: 'parse_error' };
|
|
322
|
+
return { scenario: null, reason: 'parse_error', retryable: false };
|
|
283
323
|
}
|
|
284
324
|
return { scenario };
|
|
285
325
|
} catch {
|
|
286
|
-
return { scenario: null, reason: 'parse_error' };
|
|
326
|
+
return { scenario: null, reason: 'parse_error', retryable: false };
|
|
327
|
+
}
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
async function fetchWithReason(scenarioId) {
|
|
331
|
+
if (!/^[\w-]+$/.test(scenarioId)) {
|
|
332
|
+
return { scenario: null, reason: 'invalid_id' };
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
let last;
|
|
336
|
+
for (let i = 0; i < attempts; i++) {
|
|
337
|
+
last = await attemptOnce(scenarioId);
|
|
338
|
+
if (last.scenario || !last.retryable || i === attempts - 1) break;
|
|
339
|
+
await sleep(Math.min(RETRY_BASE_MS * (1 << i), RETRY_MAX_MS));
|
|
287
340
|
}
|
|
341
|
+
// Strip the internal `retryable` flag from the public contract.
|
|
342
|
+
const { retryable: _drop, ...result } = last;
|
|
343
|
+
return result;
|
|
288
344
|
}
|
|
289
345
|
|
|
290
346
|
return {
|
package/package.json
CHANGED
|
@@ -1,13 +1,15 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dogfood-lab/ingest",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Ingestion pipeline for testing-os. Thin glue: dispatch → verifier → persist → indexes.",
|
|
6
6
|
"main": "run.js",
|
|
7
7
|
"exports": {
|
|
8
8
|
".": "./run.js",
|
|
9
9
|
"./lib/*": "./lib/*",
|
|
10
|
-
"./
|
|
10
|
+
"./anchor/*": "./anchor/*",
|
|
11
|
+
"./validate-record.js": "./validate-record.js",
|
|
12
|
+
"./verify-chain.js": "./verify-chain.js"
|
|
11
13
|
},
|
|
12
14
|
"scripts": {
|
|
13
15
|
"test": "node --test",
|
|
@@ -18,8 +20,10 @@
|
|
|
18
20
|
"persist.js",
|
|
19
21
|
"rebuild-indexes.js",
|
|
20
22
|
"validate-record.js",
|
|
23
|
+
"verify-chain.js",
|
|
21
24
|
"load-context.js",
|
|
22
25
|
"lib/",
|
|
26
|
+
"anchor/",
|
|
23
27
|
"README.md",
|
|
24
28
|
"LICENSE"
|
|
25
29
|
],
|
package/persist.js
CHANGED
|
@@ -7,11 +7,13 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { existsSync, mkdirSync, writeFileSync, renameSync, openSync, closeSync, unlinkSync } from 'node:fs';
|
|
10
|
-
import { join, dirname } from 'node:path';
|
|
10
|
+
import { join, dirname, relative, sep } from 'node:path';
|
|
11
11
|
import { randomBytes } from 'node:crypto';
|
|
12
12
|
|
|
13
13
|
import { validateRecord } from './validate-record.js';
|
|
14
14
|
import { isUnsafeSegment } from './lib/unsafe-segment.js';
|
|
15
|
+
import { submissionDigest } from './lib/integrity.js';
|
|
16
|
+
import { readChainHead, appendChainEntry } from './lib/chain-manifest.js';
|
|
15
17
|
|
|
16
18
|
/**
|
|
17
19
|
* Error thrown when writeRecord loses a TOCTOU race for the same canonical path.
|
|
@@ -125,6 +127,26 @@ export function writeRecord(record, repoRoot) {
|
|
|
125
127
|
return { path, written: false };
|
|
126
128
|
}
|
|
127
129
|
|
|
130
|
+
// Integrity chain v1 — stamp the tamper-evident integrity block BEFORE
|
|
131
|
+
// validating + writing, so the persisted record self-certifies.
|
|
132
|
+
//
|
|
133
|
+
// Serialized-ingest assumption (LOCKED CONTRACT step 4): ingest.yml is
|
|
134
|
+
// concurrency-serialized and writeRecord is synchronous, so reading the chain
|
|
135
|
+
// head and then appending after the write is race-free. A fork for
|
|
136
|
+
// truly-concurrent ingest (two writers assigning the same seq) is OUT OF SCOPE
|
|
137
|
+
// — see lib/chain-manifest.js. The order is: read head → compute digest over
|
|
138
|
+
// the record WITHOUT integrity (stable regardless of the block) → stamp
|
|
139
|
+
// integrity → write the record → append the manifest line. The append happens
|
|
140
|
+
// ONLY after the record write succeeds, and only on this real-write path
|
|
141
|
+
// (never on the duplicate short-circuit above).
|
|
142
|
+
const head = readChainHead(repoRoot);
|
|
143
|
+
const digest = submissionDigest(record);
|
|
144
|
+
record.integrity = {
|
|
145
|
+
submission_digest: digest,
|
|
146
|
+
prev_digest: head.submission_digest,
|
|
147
|
+
seq: head.seq + 1,
|
|
148
|
+
};
|
|
149
|
+
|
|
128
150
|
// Enforce dogfood-record.schema.json BEFORE touching the filesystem.
|
|
129
151
|
// Better to throw loudly than silently persist a malformed record — the
|
|
130
152
|
// schema is the contract every downstream consumer relies on.
|
|
@@ -168,5 +190,23 @@ export function writeRecord(record, repoRoot) {
|
|
|
168
190
|
throw err;
|
|
169
191
|
}
|
|
170
192
|
|
|
193
|
+
// Append the chain ledger line AFTER the record write succeeds. The manifest
|
|
194
|
+
// append is atomic (temp+rename rewrite — see lib/chain-manifest.js); a torn
|
|
195
|
+
// append cannot leave a half-line. `path` field is the record path RELATIVE to
|
|
196
|
+
// repoRoot, forward-slashed, so the ledger is portable across OSes and a line
|
|
197
|
+
// copy-pasted into a raw.githubusercontent URL is not a broken link (mirrors
|
|
198
|
+
// the posixify-at-the-boundary doctrine in run.js / rebuild-indexes.js).
|
|
199
|
+
const relPath = relative(repoRoot, path).split(sep).join('/');
|
|
200
|
+
appendChainEntry(repoRoot, {
|
|
201
|
+
seq: record.integrity.seq,
|
|
202
|
+
run_id: record.run_id,
|
|
203
|
+
repo: record.repo,
|
|
204
|
+
status: record.verification?.status ?? 'accepted',
|
|
205
|
+
path: relPath,
|
|
206
|
+
submission_digest: record.integrity.submission_digest,
|
|
207
|
+
prev_digest: record.integrity.prev_digest,
|
|
208
|
+
persisted_at: new Date().toISOString(),
|
|
209
|
+
});
|
|
210
|
+
|
|
171
211
|
return { path, written: true };
|
|
172
212
|
}
|
package/rebuild-indexes.js
CHANGED
|
@@ -8,7 +8,8 @@
|
|
|
8
8
|
*
|
|
9
9
|
* Regenerated on every accepted/rejected write in Phase 1.
|
|
10
10
|
*
|
|
11
|
-
* Multi-file commit-group
|
|
11
|
+
* Multi-file commit-group: crash/IO-failure RECOVERY-atomic, NOT reader-atomic
|
|
12
|
+
* (W3-PIPE-002):
|
|
12
13
|
* The 3 indexes are written together via a two-phase commit pattern. Phase 1
|
|
13
14
|
* stages all 3 files to temp paths AND records them in a journal file. Phase 2
|
|
14
15
|
* renames each temp into its final location, then deletes the journal. If the
|
|
@@ -17,10 +18,27 @@
|
|
|
17
18
|
* is idempotent (it scans records/ end-to-end), so re-running is the correct
|
|
18
19
|
* recovery action.
|
|
19
20
|
*
|
|
20
|
-
*
|
|
21
|
+
* IMPORTANT — what "atomicity" means here. The guarantee is RECOVERY-atomic,
|
|
22
|
+
* not READER-atomic. Phase 2 promotes the temps with a per-leg `renameSync`
|
|
23
|
+
* (each rename is individually atomic), but the GROUP is not promoted under a
|
|
24
|
+
* single atomic operation. During the promote window — and during the heal
|
|
25
|
+
* window after a mid-promote IO failure (ENOSPC/EACCES after the first
|
|
26
|
+
* final is renamed but a later one is not) — a concurrent reader CAN observe
|
|
27
|
+
* the index group in a mutually-inconsistent intermediate state (e.g. an
|
|
28
|
+
* already-promoted latest-by-repo.json against a not-yet-promoted failing.json).
|
|
29
|
+
* The catch on a promote failure does NOT roll back already-promoted finals;
|
|
30
|
+
* it preserves the journal and emits a structured error event so an operator
|
|
31
|
+
* can force an immediate rebuild before the next scheduled run heals it. The
|
|
32
|
+
* design is sound because the only writer (the ingest pipeline) serializes
|
|
33
|
+
* rebuilds and `rebuildIndexes` is synchronous — there is no in-flight reader
|
|
34
|
+
* that races a writer mid-promote within a single process. If you ever need
|
|
35
|
+
* true reader-atomicity (a reader that NEVER sees a torn group), this design
|
|
36
|
+
* must change (e.g. swap a single directory symlink, or version the index dir).
|
|
37
|
+
*
|
|
38
|
+
* Pattern reference: choke-point fix (Pattern #4) for multi-file recovery.
|
|
21
39
|
* Single-file `atomicWriteFileSync` (lib/atomic-write.js) handles each leg;
|
|
22
|
-
* the journal handles the cross-file boundary. The single-file helper
|
|
23
|
-
* the same one Class #6 helper-adoption-sweep enforces as canonical for
|
|
40
|
+
* the journal handles the cross-file recovery boundary. The single-file helper
|
|
41
|
+
* is the same one Class #6 helper-adoption-sweep enforces as canonical for
|
|
24
42
|
* temp+rename writes under `packages/ingest/`.
|
|
25
43
|
*/
|
|
26
44
|
|
|
@@ -98,6 +116,72 @@ export function findJsonFiles(dir) {
|
|
|
98
116
|
return results;
|
|
99
117
|
}
|
|
100
118
|
|
|
119
|
+
/**
|
|
120
|
+
* Probe whether `dir` exists but is unreadable at its OWN level (EACCES /
|
|
121
|
+
* Windows lock / ENOTDIR) — as distinct from a deep leaf failing mid-walk.
|
|
122
|
+
*
|
|
123
|
+
* ingest-B-002: `findJsonFiles` deliberately degrades an unreadable subtree to
|
|
124
|
+
* "those records are missing" and returns `[]`. That is correct for a single
|
|
125
|
+
* locked LEAF, but catastrophic for the records/ ROOT: a transiently-locked
|
|
126
|
+
* root makes the WHOLE corpus invisible, and an unguarded rebuild would then
|
|
127
|
+
* overwrite every index with empty content. This probe lets `rebuildIndexes`
|
|
128
|
+
* tell the two apart so it can REFUSE to clobber good indexes when the root
|
|
129
|
+
* itself is the thing that failed. A non-existent dir is NOT unreadable — that
|
|
130
|
+
* is the legitimate empty-corpus case, which must still rebuild empty indexes.
|
|
131
|
+
*
|
|
132
|
+
* @param {string} dir
|
|
133
|
+
* @returns {{ unreadable: boolean, code: string|null, error: string|null }}
|
|
134
|
+
*/
|
|
135
|
+
function probeDirReadable(dir) {
|
|
136
|
+
if (!existsSync(dir)) return { unreadable: false, code: null, error: null };
|
|
137
|
+
try {
|
|
138
|
+
readdirSync(dir);
|
|
139
|
+
return { unreadable: false, code: null, error: null };
|
|
140
|
+
} catch (err) {
|
|
141
|
+
return {
|
|
142
|
+
unreadable: true,
|
|
143
|
+
code: err && err.code ? err.code : 'readdir_failed',
|
|
144
|
+
error: err && err.message ? err.message : String(err),
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Read the prior committed latest-by-repo.json so a rebuild can tell whether
|
|
151
|
+
* the index it is about to overwrite currently has content. Used by the
|
|
152
|
+
* ingest-B-002 refuse-to-overwrite guard: an empty scan is only suspicious if
|
|
153
|
+
* the prior index was non-empty. A missing or unparseable prior index counts
|
|
154
|
+
* as "no prior content" (the legitimate first-run / empty-corpus case).
|
|
155
|
+
*
|
|
156
|
+
* @param {string} latestPath
|
|
157
|
+
* @returns {boolean} true if the prior index existed and held at least one repo
|
|
158
|
+
*/
|
|
159
|
+
function priorIndexHasContent(latestPath) {
|
|
160
|
+
if (!existsSync(latestPath)) return false;
|
|
161
|
+
try {
|
|
162
|
+
const prior = JSON.parse(readFileSync(latestPath, 'utf-8'));
|
|
163
|
+
return prior && typeof prior === 'object' && Object.keys(prior).length > 0;
|
|
164
|
+
} catch {
|
|
165
|
+
// Unparseable prior index — treat as no usable content so a corrupt index
|
|
166
|
+
// never wedges the rebuild into a permanent refuse state.
|
|
167
|
+
return false;
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* Whether the freshly-built latest-by-repo map has no repos. Used by the
|
|
173
|
+
* ingest-B-002 refuse-to-overwrite guard to recognise an empty scan. A scan
|
|
174
|
+
* can be empty because there are genuinely no accepted records (legitimate)
|
|
175
|
+
* or because the corpus was invisible (a transiently-locked records tree) —
|
|
176
|
+
* the guard combines this with `priorIndexHasContent` to tell them apart.
|
|
177
|
+
*
|
|
178
|
+
* @param {object} latestByRepo
|
|
179
|
+
* @returns {boolean}
|
|
180
|
+
*/
|
|
181
|
+
function latestByRepoIsEmpty(latestByRepo) {
|
|
182
|
+
return !latestByRepo || Object.keys(latestByRepo).length === 0;
|
|
183
|
+
}
|
|
184
|
+
|
|
101
185
|
/**
|
|
102
186
|
* Load and parse a record file.
|
|
103
187
|
*
|
|
@@ -131,8 +215,20 @@ export function rebuildIndexes(repoRoot, options = {}) {
|
|
|
131
215
|
const indexDir = join(repoRoot, 'indexes');
|
|
132
216
|
mkdirSync(indexDir, { recursive: true });
|
|
133
217
|
|
|
134
|
-
//
|
|
218
|
+
// ingest-B-002: capture two facts BEFORE the scan so we can refuse to clobber
|
|
219
|
+
// good indexes with empty ones when the corpus is invisible rather than empty.
|
|
220
|
+
// 1. Is the records/ ROOT itself unreadable (vs a deep leaf, vs absent)?
|
|
221
|
+
// A locked root makes the WHOLE corpus invisible — findJsonFiles would
|
|
222
|
+
// return [] after one low `dir_unreadable` warn, and an unguarded
|
|
223
|
+
// commit-group would then overwrite every index with {}.
|
|
224
|
+
// 2. Did the PRIOR latest-by-repo.json have content? An empty scan is only
|
|
225
|
+
// suspicious if there was something to lose; a legitimately empty
|
|
226
|
+
// first-run corpus must still write empty indexes.
|
|
135
227
|
const recordsDir = join(repoRoot, 'records');
|
|
228
|
+
const latestPath = join(indexDir, 'latest-by-repo.json');
|
|
229
|
+
const rootProbe = probeDirReadable(recordsDir);
|
|
230
|
+
const hadPriorIndex = priorIndexHasContent(latestPath);
|
|
231
|
+
|
|
136
232
|
const acceptedFiles = findJsonFiles(recordsDir)
|
|
137
233
|
.filter(f => {
|
|
138
234
|
const rel = relative(recordsDir, f);
|
|
@@ -268,9 +364,49 @@ export function rebuildIndexes(repoRoot, options = {}) {
|
|
|
268
364
|
}
|
|
269
365
|
}
|
|
270
366
|
|
|
367
|
+
// ingest-B-002: REFUSE to overwrite good indexes with empty ones when the
|
|
368
|
+
// corpus was invisible rather than genuinely empty. Two refuse conditions:
|
|
369
|
+
// - records_root_unreadable: the records/ ROOT itself failed to read
|
|
370
|
+
// (EACCES / Windows lock / ENOTDIR). The entire corpus is invisible —
|
|
371
|
+
// committing now would wipe every index. This is distinct from a single
|
|
372
|
+
// locked leaf, which findJsonFiles already degrades to a partial scan.
|
|
373
|
+
// - empty_scan_with_prior_index: the root read fine but the scan found
|
|
374
|
+
// zero accepted records while the prior latest-by-repo had content. The
|
|
375
|
+
// records likely vanished transiently; clobbering loses the portfolio.
|
|
376
|
+
// A legitimately empty corpus (no accepted records AND no prior content) is
|
|
377
|
+
// NOT refused — it must still write empty indexes (first-run case). We skip
|
|
378
|
+
// the commit-group and emit a structured, greppable event so the operator
|
|
379
|
+
// sees the refusal loudly instead of a silently-emptied portfolio.
|
|
380
|
+
const noAcceptedScanned = latestByRepoIsEmpty(latestByRepo);
|
|
381
|
+
if (rootProbe.unreadable || (noAcceptedScanned && hadPriorIndex)) {
|
|
382
|
+
const reason = rootProbe.unreadable
|
|
383
|
+
? 'records_root_unreadable'
|
|
384
|
+
: 'empty_scan_with_prior_index';
|
|
385
|
+
// A root IO failure is an operator-actionable error (the corpus is gone);
|
|
386
|
+
// an empty scan over a readable root is a warn (recoverable next run).
|
|
387
|
+
logStage(rootProbe.unreadable ? 'error' : 'warn', {
|
|
388
|
+
kind: 'index_rebuild_skipped',
|
|
389
|
+
reason,
|
|
390
|
+
records_dir: recordsDir,
|
|
391
|
+
accepted_scanned: acceptedFiles.length,
|
|
392
|
+
prior_index_non_empty: hadPriorIndex,
|
|
393
|
+
root_error_code: rootProbe.code,
|
|
394
|
+
error: rootProbe.error,
|
|
395
|
+
});
|
|
396
|
+
return {
|
|
397
|
+
latestByRepo,
|
|
398
|
+
failing,
|
|
399
|
+
stale,
|
|
400
|
+
accepted: acceptedFiles.length,
|
|
401
|
+
rejected: rejectedFiles.length,
|
|
402
|
+
corrupted,
|
|
403
|
+
skipped,
|
|
404
|
+
skippedCommit: reason,
|
|
405
|
+
};
|
|
406
|
+
}
|
|
407
|
+
|
|
271
408
|
// Write indexes via commit-group two-phase commit. See module header
|
|
272
409
|
// for the full design rationale.
|
|
273
|
-
const latestPath = join(indexDir, 'latest-by-repo.json');
|
|
274
410
|
const failingPath = join(indexDir, 'failing.json');
|
|
275
411
|
const stalePath = join(indexDir, 'stale.json');
|
|
276
412
|
|
|
@@ -307,15 +443,23 @@ export function rebuildIndexes(repoRoot, options = {}) {
|
|
|
307
443
|
* AND records them in a journal first; then renames them in caller-given
|
|
308
444
|
* order. The journal is deleted only after every rename succeeds.
|
|
309
445
|
*
|
|
310
|
-
* Crash semantics:
|
|
311
|
-
* -
|
|
446
|
+
* Crash / IO-failure semantics (RECOVERY-atomic, not reader-atomic):
|
|
447
|
+
* - Failure during STAGE phase: every staged temp is unlinked in the catch
|
|
312
448
|
* block; the journal (if written) is unlinked too. No partial visible
|
|
313
|
-
* state.
|
|
314
|
-
* -
|
|
315
|
-
* final path; remaining temps are still next to
|
|
316
|
-
*
|
|
317
|
-
*
|
|
318
|
-
*
|
|
449
|
+
* state — no final was touched.
|
|
450
|
+
* - Failure during PROMOTE phase: any successfully-renamed file is at its
|
|
451
|
+
* final path with its NEW content; remaining temps are still next to
|
|
452
|
+
* their (still-OLD) finals. The group is therefore mutually inconsistent
|
|
453
|
+
* until healed — a reader in this window sees a torn group. We do NOT
|
|
454
|
+
* roll back the already-promoted finals (their prior content was already
|
|
455
|
+
* overwritten by the atomic rename — there is nothing to roll back to
|
|
456
|
+
* without re-reading the journal). Instead we emit a structured
|
|
457
|
+
* `logStage('error', { kind: 'commit_group_partial_promote', ... })`
|
|
458
|
+
* naming which finals were promoted vs left stale so an operator can
|
|
459
|
+
* force an immediate rebuild, and we preserve the journal. Next run's
|
|
460
|
+
* `cleanupCrashedJournals` deletes residual temps and the journal; the
|
|
461
|
+
* next normal `rebuildIndexes` call rewrites all 3 indexes from scratch
|
|
462
|
+
* (idempotent), which is what heals the torn group.
|
|
319
463
|
*
|
|
320
464
|
* Why journal-then-rename rather than journal-only: the rename phase needs
|
|
321
465
|
* to be the visible commit point. A journal-only design would require
|
|
@@ -339,8 +483,12 @@ function commitGroupRename(indexDir, entries) {
|
|
|
339
483
|
|
|
340
484
|
// Write journal AFTER staging so it never points at a non-existent temp.
|
|
341
485
|
// Atomic write of the journal itself: writeFileSync directly is fine here
|
|
342
|
-
// because the journal is process-private
|
|
343
|
-
// collision
|
|
486
|
+
// because the journal is process-private — the pid + random suffix make
|
|
487
|
+
// the filename collision-free, and `cleanupCrashedJournals` is pid-aware
|
|
488
|
+
// (it skips journals whose pid is a still-live process), so a future
|
|
489
|
+
// concurrent rebuild's in-flight journal is never reaped out from under
|
|
490
|
+
// it. The temp `entries` it lists are equally collision-free (each carries
|
|
491
|
+
// its own random suffix from `stageWriteFileSync`).
|
|
344
492
|
writeFileSync(
|
|
345
493
|
journalPath,
|
|
346
494
|
JSON.stringify({
|
|
@@ -374,6 +522,24 @@ function commitGroupRename(indexDir, entries) {
|
|
|
374
522
|
// (their previous content is already overwritten — the rename was
|
|
375
523
|
// atomic at each individual leg, just not as a group). The next run
|
|
376
524
|
// is idempotent and will rewrite all three from scratch.
|
|
525
|
+
//
|
|
526
|
+
// ingest-A-001: the group is now reader-inconsistent (promoted finals
|
|
527
|
+
// carry new content; stale finals carry old content). Name which finals
|
|
528
|
+
// are which in a structured error event so an operator can force an
|
|
529
|
+
// immediate rebuild rather than wait for the next scheduled run to heal
|
|
530
|
+
// the torn group.
|
|
531
|
+
const promoted = stagedTmps.slice(0, promotedCount).map((e) => e.finalPath);
|
|
532
|
+
const stale = stagedTmps.slice(promotedCount).map((e) => e.finalPath);
|
|
533
|
+
logStage('error', {
|
|
534
|
+
kind: 'commit_group_partial_promote',
|
|
535
|
+
reason: err && err.code ? err.code : 'promote_failed',
|
|
536
|
+
promoted_count: promotedCount,
|
|
537
|
+
total: stagedTmps.length,
|
|
538
|
+
promoted_finals: promoted,
|
|
539
|
+
stale_finals: stale,
|
|
540
|
+
journal: journalPath,
|
|
541
|
+
error: err && err.message ? err.message : String(err),
|
|
542
|
+
});
|
|
377
543
|
throw new Error(
|
|
378
544
|
`commitGroupRename: promote failed after ${promotedCount}/${stagedTmps.length} files; ` +
|
|
379
545
|
`journal preserved at ${journalPath} for next-run cleanup. Original error: ${err.message}`
|
|
@@ -387,11 +553,42 @@ function commitGroupRename(indexDir, entries) {
|
|
|
387
553
|
try { unlinkSync(journalPath); } catch { /* will be cleaned next run */ }
|
|
388
554
|
}
|
|
389
555
|
|
|
556
|
+
/**
|
|
557
|
+
* Probe whether a pid is still a live process. `process.kill(pid, 0)` sends
|
|
558
|
+
* no signal — it only performs the permission/existence check, throwing
|
|
559
|
+
* ESRCH when the pid is dead. An EPERM means the process exists but is owned
|
|
560
|
+
* by another user; that still counts as "live" for our purpose (do not reap
|
|
561
|
+
* its journal). Any other error (or a non-integer pid) is treated as "not
|
|
562
|
+
* provably live" so a malformed journal never blocks its own cleanup.
|
|
563
|
+
*
|
|
564
|
+
* @param {unknown} pid
|
|
565
|
+
* @returns {boolean}
|
|
566
|
+
*/
|
|
567
|
+
function isProcessAlive(pid) {
|
|
568
|
+
if (!Number.isInteger(pid) || pid <= 0) return false;
|
|
569
|
+
try {
|
|
570
|
+
process.kill(pid, 0);
|
|
571
|
+
return true;
|
|
572
|
+
} catch (err) {
|
|
573
|
+
return err && err.code === 'EPERM';
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
|
|
390
577
|
/**
|
|
391
578
|
* Find and clean up any in-progress journals from previous runs. Each journal
|
|
392
579
|
* lists the temp paths that were staged; we unlink any that still exist
|
|
393
580
|
* (they are residue from a crashed run) and delete the journal.
|
|
394
581
|
*
|
|
582
|
+
* ingest-A-002: cleanup is PID-AWARE. A journal whose `pid` is a still-live
|
|
583
|
+
* process is the in-flight recovery state of a concurrent rebuild — reaping
|
|
584
|
+
* it would delete that run's temps and journal mid-flight. Today the only
|
|
585
|
+
* writer serializes rebuilds and `rebuildIndexes` is synchronous, so no live
|
|
586
|
+
* sibling journal exists at Phase-0 cleanup time; this guard makes the design
|
|
587
|
+
* correct (not merely safe-by-serialization) so a future maintainer who adds
|
|
588
|
+
* concurrency does not silently corrupt a peer. A dead pid, a missing/
|
|
589
|
+
* malformed pid, or an unreadable journal is still reaped — that is the
|
|
590
|
+
* crashed-run residue this function exists to clear.
|
|
591
|
+
*
|
|
395
592
|
* Idempotent: on a clean filesystem it's a no-op; on a crashed-mid-promote
|
|
396
593
|
* filesystem it cleans the slate so the upcoming `commitGroupRename` can
|
|
397
594
|
* stage fresh temps without colliding.
|
|
@@ -412,6 +609,11 @@ function cleanupCrashedJournals(indexDir) {
|
|
|
412
609
|
// referenced will linger but they're harmless (they have a unique
|
|
413
610
|
// suffix that won't be re-used).
|
|
414
611
|
}
|
|
612
|
+
// Skip a journal owned by a still-live process — it belongs to a
|
|
613
|
+
// concurrent rebuild's in-flight recovery state, not crashed residue.
|
|
614
|
+
if (parsed && isProcessAlive(parsed.pid) && parsed.pid !== process.pid) {
|
|
615
|
+
continue;
|
|
616
|
+
}
|
|
415
617
|
if (parsed && Array.isArray(parsed.entries)) {
|
|
416
618
|
for (const e of parsed.entries) {
|
|
417
619
|
if (e && typeof e.tmpPath === 'string') {
|