@dogfood-lab/ingest 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -22
- package/anchor/cli.js +185 -0
- package/anchor/compute-root.js +302 -0
- package/anchor/config.js +44 -0
- package/anchor/merkle.js +123 -0
- package/anchor/post-anchor.js +299 -0
- package/anchor/verify-anchor.js +358 -0
- package/lib/chain-manifest.js +106 -0
- package/lib/integrity.js +98 -0
- package/package.json +6 -2
- package/persist.js +41 -1
- package/rebuild-indexes.js +218 -16
- package/run.js +160 -10
- package/verify-chain.js +178 -0
package/rebuild-indexes.js
CHANGED
|
@@ -8,7 +8,8 @@
|
|
|
8
8
|
*
|
|
9
9
|
* Regenerated on every accepted/rejected write in Phase 1.
|
|
10
10
|
*
|
|
11
|
-
* Multi-file commit-group
|
|
11
|
+
* Multi-file commit-group: crash/IO-failure RECOVERY-atomic, NOT reader-atomic
|
|
12
|
+
* (W3-PIPE-002):
|
|
12
13
|
* The 3 indexes are written together via a two-phase commit pattern. Phase 1
|
|
13
14
|
* stages all 3 files to temp paths AND records them in a journal file. Phase 2
|
|
14
15
|
* renames each temp into its final location, then deletes the journal. If the
|
|
@@ -17,10 +18,27 @@
|
|
|
17
18
|
* is idempotent (it scans records/ end-to-end), so re-running is the correct
|
|
18
19
|
* recovery action.
|
|
19
20
|
*
|
|
20
|
-
*
|
|
21
|
+
* IMPORTANT — what "atomicity" means here. The guarantee is RECOVERY-atomic,
|
|
22
|
+
* not READER-atomic. Phase 2 promotes the temps with a per-leg `renameSync`
|
|
23
|
+
* (each rename is individually atomic), but the GROUP is not promoted under a
|
|
24
|
+
* single atomic operation. During the promote window — and during the heal
|
|
25
|
+
* window after a mid-promote IO failure (ENOSPC/EACCES after the first
|
|
26
|
+
* final is renamed but a later one is not) — a concurrent reader CAN observe
|
|
27
|
+
* the index group in a mutually-inconsistent intermediate state (e.g. an
|
|
28
|
+
* already-promoted latest-by-repo.json against a not-yet-promoted failing.json).
|
|
29
|
+
* The catch on a promote failure does NOT roll back already-promoted finals;
|
|
30
|
+
* it preserves the journal and emits a structured error event so an operator
|
|
31
|
+
* can force an immediate rebuild before the next scheduled run heals it. The
|
|
32
|
+
* design is sound because the only writer (the ingest pipeline) serializes
|
|
33
|
+
* rebuilds and `rebuildIndexes` is synchronous — there is no in-flight reader
|
|
34
|
+
* that races a writer mid-promote within a single process. If you ever need
|
|
35
|
+
* true reader-atomicity (a reader that NEVER sees a torn group), this design
|
|
36
|
+
* must change (e.g. swap a single directory symlink, or version the index dir).
|
|
37
|
+
*
|
|
38
|
+
* Pattern reference: choke-point fix (Pattern #4) for multi-file recovery.
|
|
21
39
|
* Single-file `atomicWriteFileSync` (lib/atomic-write.js) handles each leg;
|
|
22
|
-
* the journal handles the cross-file boundary. The single-file helper
|
|
23
|
-
* the same one Class #6 helper-adoption-sweep enforces as canonical for
|
|
40
|
+
* the journal handles the cross-file recovery boundary. The single-file helper
|
|
41
|
+
* is the same one Class #6 helper-adoption-sweep enforces as canonical for
|
|
24
42
|
* temp+rename writes under `packages/ingest/`.
|
|
25
43
|
*/
|
|
26
44
|
|
|
@@ -98,6 +116,72 @@ export function findJsonFiles(dir) {
|
|
|
98
116
|
return results;
|
|
99
117
|
}
|
|
100
118
|
|
|
119
|
+
/**
|
|
120
|
+
* Probe whether `dir` exists but is unreadable at its OWN level (EACCES /
|
|
121
|
+
* Windows lock / ENOTDIR) — as distinct from a deep leaf failing mid-walk.
|
|
122
|
+
*
|
|
123
|
+
* ingest-B-002: `findJsonFiles` deliberately degrades an unreadable subtree to
|
|
124
|
+
* "those records are missing" and returns `[]`. That is correct for a single
|
|
125
|
+
* locked LEAF, but catastrophic for the records/ ROOT: a transiently-locked
|
|
126
|
+
* root makes the WHOLE corpus invisible, and an unguarded rebuild would then
|
|
127
|
+
* overwrite every index with empty content. This probe lets `rebuildIndexes`
|
|
128
|
+
* tell the two apart so it can REFUSE to clobber good indexes when the root
|
|
129
|
+
* itself is the thing that failed. A non-existent dir is NOT unreadable — that
|
|
130
|
+
* is the legitimate empty-corpus case, which must still rebuild empty indexes.
|
|
131
|
+
*
|
|
132
|
+
* @param {string} dir
|
|
133
|
+
* @returns {{ unreadable: boolean, code: string|null, error: string|null }}
|
|
134
|
+
*/
|
|
135
|
+
function probeDirReadable(dir) {
|
|
136
|
+
if (!existsSync(dir)) return { unreadable: false, code: null, error: null };
|
|
137
|
+
try {
|
|
138
|
+
readdirSync(dir);
|
|
139
|
+
return { unreadable: false, code: null, error: null };
|
|
140
|
+
} catch (err) {
|
|
141
|
+
return {
|
|
142
|
+
unreadable: true,
|
|
143
|
+
code: err && err.code ? err.code : 'readdir_failed',
|
|
144
|
+
error: err && err.message ? err.message : String(err),
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Read the prior committed latest-by-repo.json so a rebuild can tell whether
|
|
151
|
+
* the index it is about to overwrite currently has content. Used by the
|
|
152
|
+
* ingest-B-002 refuse-to-overwrite guard: an empty scan is only suspicious if
|
|
153
|
+
* the prior index was non-empty. A missing or unparseable prior index counts
|
|
154
|
+
* as "no prior content" (the legitimate first-run / empty-corpus case).
|
|
155
|
+
*
|
|
156
|
+
* @param {string} latestPath
|
|
157
|
+
* @returns {boolean} true if the prior index existed and held at least one repo
|
|
158
|
+
*/
|
|
159
|
+
function priorIndexHasContent(latestPath) {
|
|
160
|
+
if (!existsSync(latestPath)) return false;
|
|
161
|
+
try {
|
|
162
|
+
const prior = JSON.parse(readFileSync(latestPath, 'utf-8'));
|
|
163
|
+
return prior && typeof prior === 'object' && Object.keys(prior).length > 0;
|
|
164
|
+
} catch {
|
|
165
|
+
// Unparseable prior index — treat as no usable content so a corrupt index
|
|
166
|
+
// never wedges the rebuild into a permanent refuse state.
|
|
167
|
+
return false;
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* Whether the freshly-built latest-by-repo map has no repos. Used by the
|
|
173
|
+
* ingest-B-002 refuse-to-overwrite guard to recognise an empty scan. A scan
|
|
174
|
+
* can be empty because there are genuinely no accepted records (legitimate)
|
|
175
|
+
* or because the corpus was invisible (a transiently-locked records tree) —
|
|
176
|
+
* the guard combines this with `priorIndexHasContent` to tell them apart.
|
|
177
|
+
*
|
|
178
|
+
* @param {object} latestByRepo
|
|
179
|
+
* @returns {boolean}
|
|
180
|
+
*/
|
|
181
|
+
function latestByRepoIsEmpty(latestByRepo) {
|
|
182
|
+
return !latestByRepo || Object.keys(latestByRepo).length === 0;
|
|
183
|
+
}
|
|
184
|
+
|
|
101
185
|
/**
|
|
102
186
|
* Load and parse a record file.
|
|
103
187
|
*
|
|
@@ -131,8 +215,20 @@ export function rebuildIndexes(repoRoot, options = {}) {
|
|
|
131
215
|
const indexDir = join(repoRoot, 'indexes');
|
|
132
216
|
mkdirSync(indexDir, { recursive: true });
|
|
133
217
|
|
|
134
|
-
//
|
|
218
|
+
// ingest-B-002: capture two facts BEFORE the scan so we can refuse to clobber
|
|
219
|
+
// good indexes with empty ones when the corpus is invisible rather than empty.
|
|
220
|
+
// 1. Is the records/ ROOT itself unreadable (vs a deep leaf, vs absent)?
|
|
221
|
+
// A locked root makes the WHOLE corpus invisible — findJsonFiles would
|
|
222
|
+
// return [] after one low `dir_unreadable` warn, and an unguarded
|
|
223
|
+
// commit-group would then overwrite every index with {}.
|
|
224
|
+
// 2. Did the PRIOR latest-by-repo.json have content? An empty scan is only
|
|
225
|
+
// suspicious if there was something to lose; a legitimately empty
|
|
226
|
+
// first-run corpus must still write empty indexes.
|
|
135
227
|
const recordsDir = join(repoRoot, 'records');
|
|
228
|
+
const latestPath = join(indexDir, 'latest-by-repo.json');
|
|
229
|
+
const rootProbe = probeDirReadable(recordsDir);
|
|
230
|
+
const hadPriorIndex = priorIndexHasContent(latestPath);
|
|
231
|
+
|
|
136
232
|
const acceptedFiles = findJsonFiles(recordsDir)
|
|
137
233
|
.filter(f => {
|
|
138
234
|
const rel = relative(recordsDir, f);
|
|
@@ -268,9 +364,49 @@ export function rebuildIndexes(repoRoot, options = {}) {
|
|
|
268
364
|
}
|
|
269
365
|
}
|
|
270
366
|
|
|
367
|
+
// ingest-B-002: REFUSE to overwrite good indexes with empty ones when the
|
|
368
|
+
// corpus was invisible rather than genuinely empty. Two refuse conditions:
|
|
369
|
+
// - records_root_unreadable: the records/ ROOT itself failed to read
|
|
370
|
+
// (EACCES / Windows lock / ENOTDIR). The entire corpus is invisible —
|
|
371
|
+
// committing now would wipe every index. This is distinct from a single
|
|
372
|
+
// locked leaf, which findJsonFiles already degrades to a partial scan.
|
|
373
|
+
// - empty_scan_with_prior_index: the root read fine but the scan found
|
|
374
|
+
// zero accepted records while the prior latest-by-repo had content. The
|
|
375
|
+
// records likely vanished transiently; clobbering loses the portfolio.
|
|
376
|
+
// A legitimately empty corpus (no accepted records AND no prior content) is
|
|
377
|
+
// NOT refused — it must still write empty indexes (first-run case). We skip
|
|
378
|
+
// the commit-group and emit a structured, greppable event so the operator
|
|
379
|
+
// sees the refusal loudly instead of a silently-emptied portfolio.
|
|
380
|
+
const noAcceptedScanned = latestByRepoIsEmpty(latestByRepo);
|
|
381
|
+
if (rootProbe.unreadable || (noAcceptedScanned && hadPriorIndex)) {
|
|
382
|
+
const reason = rootProbe.unreadable
|
|
383
|
+
? 'records_root_unreadable'
|
|
384
|
+
: 'empty_scan_with_prior_index';
|
|
385
|
+
// A root IO failure is an operator-actionable error (the corpus is gone);
|
|
386
|
+
// an empty scan over a readable root is a warn (recoverable next run).
|
|
387
|
+
logStage(rootProbe.unreadable ? 'error' : 'warn', {
|
|
388
|
+
kind: 'index_rebuild_skipped',
|
|
389
|
+
reason,
|
|
390
|
+
records_dir: recordsDir,
|
|
391
|
+
accepted_scanned: acceptedFiles.length,
|
|
392
|
+
prior_index_non_empty: hadPriorIndex,
|
|
393
|
+
root_error_code: rootProbe.code,
|
|
394
|
+
error: rootProbe.error,
|
|
395
|
+
});
|
|
396
|
+
return {
|
|
397
|
+
latestByRepo,
|
|
398
|
+
failing,
|
|
399
|
+
stale,
|
|
400
|
+
accepted: acceptedFiles.length,
|
|
401
|
+
rejected: rejectedFiles.length,
|
|
402
|
+
corrupted,
|
|
403
|
+
skipped,
|
|
404
|
+
skippedCommit: reason,
|
|
405
|
+
};
|
|
406
|
+
}
|
|
407
|
+
|
|
271
408
|
// Write indexes via commit-group two-phase commit. See module header
|
|
272
409
|
// for the full design rationale.
|
|
273
|
-
const latestPath = join(indexDir, 'latest-by-repo.json');
|
|
274
410
|
const failingPath = join(indexDir, 'failing.json');
|
|
275
411
|
const stalePath = join(indexDir, 'stale.json');
|
|
276
412
|
|
|
@@ -307,15 +443,23 @@ export function rebuildIndexes(repoRoot, options = {}) {
|
|
|
307
443
|
* AND records them in a journal first; then renames them in caller-given
|
|
308
444
|
* order. The journal is deleted only after every rename succeeds.
|
|
309
445
|
*
|
|
310
|
-
* Crash semantics:
|
|
311
|
-
* -
|
|
446
|
+
* Crash / IO-failure semantics (RECOVERY-atomic, not reader-atomic):
|
|
447
|
+
* - Failure during STAGE phase: every staged temp is unlinked in the catch
|
|
312
448
|
* block; the journal (if written) is unlinked too. No partial visible
|
|
313
|
-
* state.
|
|
314
|
-
* -
|
|
315
|
-
* final path; remaining temps are still next to
|
|
316
|
-
*
|
|
317
|
-
*
|
|
318
|
-
*
|
|
449
|
+
* state — no final was touched.
|
|
450
|
+
* - Failure during PROMOTE phase: any successfully-renamed file is at its
|
|
451
|
+
* final path with its NEW content; remaining temps are still next to
|
|
452
|
+
* their (still-OLD) finals. The group is therefore mutually inconsistent
|
|
453
|
+
* until healed — a reader in this window sees a torn group. We do NOT
|
|
454
|
+
* roll back the already-promoted finals (their prior content was already
|
|
455
|
+
* overwritten by the atomic rename — there is nothing to roll back to
|
|
456
|
+
* without re-reading the journal). Instead we emit a structured
|
|
457
|
+
* `logStage('error', { kind: 'commit_group_partial_promote', ... })`
|
|
458
|
+
* naming which finals were promoted vs left stale so an operator can
|
|
459
|
+
* force an immediate rebuild, and we preserve the journal. Next run's
|
|
460
|
+
* `cleanupCrashedJournals` deletes residual temps and the journal; the
|
|
461
|
+
* next normal `rebuildIndexes` call rewrites all 3 indexes from scratch
|
|
462
|
+
* (idempotent), which is what heals the torn group.
|
|
319
463
|
*
|
|
320
464
|
* Why journal-then-rename rather than journal-only: the rename phase needs
|
|
321
465
|
* to be the visible commit point. A journal-only design would require
|
|
@@ -339,8 +483,12 @@ function commitGroupRename(indexDir, entries) {
|
|
|
339
483
|
|
|
340
484
|
// Write journal AFTER staging so it never points at a non-existent temp.
|
|
341
485
|
// Atomic write of the journal itself: writeFileSync directly is fine here
|
|
342
|
-
// because the journal is process-private
|
|
343
|
-
// collision
|
|
486
|
+
// because the journal is process-private — the pid + random suffix make
|
|
487
|
+
// the filename collision-free, and `cleanupCrashedJournals` is pid-aware
|
|
488
|
+
// (it skips journals whose pid is a still-live process), so a future
|
|
489
|
+
// concurrent rebuild's in-flight journal is never reaped out from under
|
|
490
|
+
// it. The temp `entries` it lists are equally collision-free (each carries
|
|
491
|
+
// its own random suffix from `stageWriteFileSync`).
|
|
344
492
|
writeFileSync(
|
|
345
493
|
journalPath,
|
|
346
494
|
JSON.stringify({
|
|
@@ -374,6 +522,24 @@ function commitGroupRename(indexDir, entries) {
|
|
|
374
522
|
// (their previous content is already overwritten — the rename was
|
|
375
523
|
// atomic at each individual leg, just not as a group). The next run
|
|
376
524
|
// is idempotent and will rewrite all three from scratch.
|
|
525
|
+
//
|
|
526
|
+
// ingest-A-001: the group is now reader-inconsistent (promoted finals
|
|
527
|
+
// carry new content; stale finals carry old content). Name which finals
|
|
528
|
+
// are which in a structured error event so an operator can force an
|
|
529
|
+
// immediate rebuild rather than wait for the next scheduled run to heal
|
|
530
|
+
// the torn group.
|
|
531
|
+
const promoted = stagedTmps.slice(0, promotedCount).map((e) => e.finalPath);
|
|
532
|
+
const stale = stagedTmps.slice(promotedCount).map((e) => e.finalPath);
|
|
533
|
+
logStage('error', {
|
|
534
|
+
kind: 'commit_group_partial_promote',
|
|
535
|
+
reason: err && err.code ? err.code : 'promote_failed',
|
|
536
|
+
promoted_count: promotedCount,
|
|
537
|
+
total: stagedTmps.length,
|
|
538
|
+
promoted_finals: promoted,
|
|
539
|
+
stale_finals: stale,
|
|
540
|
+
journal: journalPath,
|
|
541
|
+
error: err && err.message ? err.message : String(err),
|
|
542
|
+
});
|
|
377
543
|
throw new Error(
|
|
378
544
|
`commitGroupRename: promote failed after ${promotedCount}/${stagedTmps.length} files; ` +
|
|
379
545
|
`journal preserved at ${journalPath} for next-run cleanup. Original error: ${err.message}`
|
|
@@ -387,11 +553,42 @@ function commitGroupRename(indexDir, entries) {
|
|
|
387
553
|
try { unlinkSync(journalPath); } catch { /* will be cleaned next run */ }
|
|
388
554
|
}
|
|
389
555
|
|
|
556
|
+
/**
|
|
557
|
+
* Probe whether a pid is still a live process. `process.kill(pid, 0)` sends
|
|
558
|
+
* no signal — it only performs the permission/existence check, throwing
|
|
559
|
+
* ESRCH when the pid is dead. An EPERM means the process exists but is owned
|
|
560
|
+
* by another user; that still counts as "live" for our purpose (do not reap
|
|
561
|
+
* its journal). Any other error (or a non-integer pid) is treated as "not
|
|
562
|
+
* provably live" so a malformed journal never blocks its own cleanup.
|
|
563
|
+
*
|
|
564
|
+
* @param {unknown} pid
|
|
565
|
+
* @returns {boolean}
|
|
566
|
+
*/
|
|
567
|
+
function isProcessAlive(pid) {
|
|
568
|
+
if (!Number.isInteger(pid) || pid <= 0) return false;
|
|
569
|
+
try {
|
|
570
|
+
process.kill(pid, 0);
|
|
571
|
+
return true;
|
|
572
|
+
} catch (err) {
|
|
573
|
+
return err && err.code === 'EPERM';
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
|
|
390
577
|
/**
|
|
391
578
|
* Find and clean up any in-progress journals from previous runs. Each journal
|
|
392
579
|
* lists the temp paths that were staged; we unlink any that still exist
|
|
393
580
|
* (they are residue from a crashed run) and delete the journal.
|
|
394
581
|
*
|
|
582
|
+
* ingest-A-002: cleanup is PID-AWARE. A journal whose `pid` is a still-live
|
|
583
|
+
* process is the in-flight recovery state of a concurrent rebuild — reaping
|
|
584
|
+
* it would delete that run's temps and journal mid-flight. Today the only
|
|
585
|
+
* writer serializes rebuilds and `rebuildIndexes` is synchronous, so no live
|
|
586
|
+
* sibling journal exists at Phase-0 cleanup time; this guard makes the design
|
|
587
|
+
* correct (not merely safe-by-serialization) so a future maintainer who adds
|
|
588
|
+
* concurrency does not silently corrupt a peer. A dead pid, a missing/
|
|
589
|
+
* malformed pid, or an unreadable journal is still reaped — that is the
|
|
590
|
+
* crashed-run residue this function exists to clear.
|
|
591
|
+
*
|
|
395
592
|
* Idempotent: on a clean filesystem it's a no-op; on a crashed-mid-promote
|
|
396
593
|
* filesystem it cleans the slate so the upcoming `commitGroupRename` can
|
|
397
594
|
* stage fresh temps without colliding.
|
|
@@ -412,6 +609,11 @@ function cleanupCrashedJournals(indexDir) {
|
|
|
412
609
|
// referenced will linger but they're harmless (they have a unique
|
|
413
610
|
// suffix that won't be re-used).
|
|
414
611
|
}
|
|
612
|
+
// Skip a journal owned by a still-live process — it belongs to a
|
|
613
|
+
// concurrent rebuild's in-flight recovery state, not crashed residue.
|
|
614
|
+
if (parsed && isProcessAlive(parsed.pid) && parsed.pid !== process.pid) {
|
|
615
|
+
continue;
|
|
616
|
+
}
|
|
415
617
|
if (parsed && Array.isArray(parsed.entries)) {
|
|
416
618
|
for (const e of parsed.entries) {
|
|
417
619
|
if (e && typeof e.tmpPath === 'string') {
|
package/run.js
CHANGED
|
@@ -22,14 +22,44 @@ import { fileURLToPath } from 'node:url';
|
|
|
22
22
|
import { randomBytes } from 'node:crypto';
|
|
23
23
|
|
|
24
24
|
import { verify } from '@dogfood-lab/verify';
|
|
25
|
-
import { stubProvenance,
|
|
25
|
+
import { stubProvenance, provenanceForProvider } from '@dogfood-lab/verify/validators/provenance.js';
|
|
26
26
|
import { logStage as sharedLogStage } from '@dogfood-lab/dogfood-swarm/lib/log-stage.js';
|
|
27
27
|
import { loadGlobalPolicy, loadRepoPolicy, loadScenarios } from './load-context.js';
|
|
28
28
|
import { isDuplicate, writeRecord, computeRecordPath } from './persist.js';
|
|
29
29
|
import { rebuildIndexes } from './rebuild-indexes.js';
|
|
30
|
+
import { verifyChain, formatChainResult } from './verify-chain.js';
|
|
31
|
+
import { handleAnchorCompute, handleAnchorPost, handleAnchorVerify } from './anchor/cli.js';
|
|
30
32
|
|
|
31
33
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
32
34
|
|
|
35
|
+
/**
|
|
36
|
+
* Resolve the REAL provenance adapter for a submission, routed by
|
|
37
|
+
* `submission.source.provider` (github | gitlab), sourcing the provider's token
|
|
38
|
+
* from the environment. Returns `{ provenance }` on success or `{ err }` (a
|
|
39
|
+
* structured, operator-legible Error the caller surfaces via emitCliErrorEvent
|
|
40
|
+
* + exit 2). The adapter registry (`provenanceForProvider`) is the single
|
|
41
|
+
* provider-keyed seam; a provider in the schema enum without a registered
|
|
42
|
+
* adapter fails here loudly rather than silently skipping verification.
|
|
43
|
+
*
|
|
44
|
+
* @param {object} submission
|
|
45
|
+
* @returns {{ provenance: object } | { err: Error }}
|
|
46
|
+
*/
|
|
47
|
+
function resolveProviderProvenance(submission) {
|
|
48
|
+
const provider = (submission && submission.source && submission.source.provider) || 'github';
|
|
49
|
+
const factory = provenanceForProvider(provider);
|
|
50
|
+
if (!factory) {
|
|
51
|
+
return { err: new Error(`unknown provenance provider '${provider}' — no adapter registered (supported: github, gitlab).`) };
|
|
52
|
+
}
|
|
53
|
+
const token = provider === 'gitlab'
|
|
54
|
+
? (process.env.GITLAB_TOKEN || process.env.CI_JOB_TOKEN)
|
|
55
|
+
: (process.env.GITHUB_TOKEN || process.env.GH_TOKEN);
|
|
56
|
+
if (!token) {
|
|
57
|
+
const need = provider === 'gitlab' ? 'GITLAB_TOKEN or CI_JOB_TOKEN' : 'GITHUB_TOKEN or GH_TOKEN';
|
|
58
|
+
return { err: new Error(`real provenance for provider '${provider}' requires ${need} in the environment.`) };
|
|
59
|
+
}
|
|
60
|
+
return { provenance: factory(token) };
|
|
61
|
+
}
|
|
62
|
+
|
|
33
63
|
/**
|
|
34
64
|
* SEED-1 (d3-ingest-003) — posixify a path-shaped value at the operator/log
|
|
35
65
|
* SERIALIZATION boundary. `computeRecordPath`/`writeRecord` return OS-native
|
|
@@ -513,6 +543,18 @@ if (isMain) {
|
|
|
513
543
|
let submissionJson;
|
|
514
544
|
let provenanceMode = null;
|
|
515
545
|
let verifyOnlyFlag = false;
|
|
546
|
+
let verifyChainFlag = false;
|
|
547
|
+
// Anchor verbs (optional, off-by-default, operator-run). --anchor-compute and
|
|
548
|
+
// --anchor-verify are fully offline (never import xrpl); --anchor-post lazily
|
|
549
|
+
// loads the optional xrpl package and needs XRPL_SEED.
|
|
550
|
+
let anchorComputeFlag = false;
|
|
551
|
+
let anchorPostFlag = false;
|
|
552
|
+
let anchorVerifyFlag = false;
|
|
553
|
+
let anchorMode = 'since-last';
|
|
554
|
+
let anchorAlgo = null;
|
|
555
|
+
let anchorNetwork = null;
|
|
556
|
+
let anchorTxFile = null;
|
|
557
|
+
let anchorTrustedAccounts = [];
|
|
516
558
|
const positionalArgs = [];
|
|
517
559
|
|
|
518
560
|
for (let i = 0; i < args.length; i++) {
|
|
@@ -563,11 +605,115 @@ if (isMain) {
|
|
|
563
605
|
// F-252714-058: dry-run the pipeline without writing or rebuilding
|
|
564
606
|
// indexes. CI / operators preview what WOULD have been persisted.
|
|
565
607
|
verifyOnlyFlag = true;
|
|
608
|
+
} else if (arg === '--verify-chain') {
|
|
609
|
+
// Integrity chain v1: verify the append-only tamper-evident ledger at
|
|
610
|
+
// indexes/integrity/chain.jsonl, fully offline. No submission, no stdin,
|
|
611
|
+
// no provenance — a standalone audit command.
|
|
612
|
+
verifyChainFlag = true;
|
|
613
|
+
} else if (arg === '--anchor-compute') {
|
|
614
|
+
// Optional XRPL anchor: compute + write the next anchor manifest. Offline.
|
|
615
|
+
anchorComputeFlag = true;
|
|
616
|
+
} else if (arg === '--anchor-post') {
|
|
617
|
+
// Optional XRPL anchor: compute if needed + post to XRPL. Needs the
|
|
618
|
+
// optional xrpl package (lazily loaded) and XRPL_SEED.
|
|
619
|
+
anchorPostFlag = true;
|
|
620
|
+
} else if (arg === '--anchor-verify') {
|
|
621
|
+
// Optional XRPL anchor: verify local manifests + run the truncation check.
|
|
622
|
+
// Offline reports honest NOT-verified for the on-chain leg.
|
|
623
|
+
anchorVerifyFlag = true;
|
|
624
|
+
} else if (arg === '--anchor-all') {
|
|
625
|
+
// Genesis snapshot mode for compute/post (covers the whole chain).
|
|
626
|
+
anchorMode = 'all';
|
|
627
|
+
} else if (arg === '--anchor-algo' && hasValue) {
|
|
628
|
+
anchorAlgo = takeValue();
|
|
629
|
+
} else if (arg === '--anchor-network' && hasValue) {
|
|
630
|
+
anchorNetwork = takeValue();
|
|
631
|
+
} else if (arg === '--anchor-tx' && hasValue) {
|
|
632
|
+
// Path to a JSON file containing a fetched XRPL tx (with Memos) for the
|
|
633
|
+
// on-chain leg of --anchor-verify. Offline-honest: omit it to run the
|
|
634
|
+
// truncation check only.
|
|
635
|
+
anchorTxFile = takeValue();
|
|
636
|
+
} else if (arg === '--anchor-trusted' && hasValue) {
|
|
637
|
+
// Comma-separated trusted anchor accounts (UNIONed with the bundled list).
|
|
638
|
+
anchorTrustedAccounts = takeValue().split(',').map((s) => s.trim()).filter(Boolean);
|
|
566
639
|
} else {
|
|
567
640
|
positionalArgs.push(args[i]);
|
|
568
641
|
}
|
|
569
642
|
}
|
|
570
643
|
|
|
644
|
+
// --verify-chain is a standalone, side-effect-free audit: it reads only the
|
|
645
|
+
// ledger + the record files it references, takes no submission, reads no
|
|
646
|
+
// stdin, and needs no provenance adapter. Handle it BEFORE the stdin read and
|
|
647
|
+
// provenance resolution so `node run.js --verify-chain` does not block on
|
|
648
|
+
// stdin or demand a --provenance flag. Exit 0 when the chain verifies, 1 on
|
|
649
|
+
// the first break (operator-legible output, no raw stack traces).
|
|
650
|
+
if (verifyChainFlag) {
|
|
651
|
+
const result = verifyChain(repoRoot);
|
|
652
|
+
logStage(result.ok ? 'verify_chain_complete' : 'error', {
|
|
653
|
+
correlation_id: synthCorrelationId(),
|
|
654
|
+
...(result.ok ? {} : { failed_stage: 'verify_chain' }),
|
|
655
|
+
verified: result.count,
|
|
656
|
+
head_digest: result.head_digest,
|
|
657
|
+
chain_ok: result.ok,
|
|
658
|
+
...(result.break ? { break_seq: result.break.seq, break_reason: result.break.reason } : {})
|
|
659
|
+
});
|
|
660
|
+
const lines = formatChainResult(result);
|
|
661
|
+
if (result.ok) {
|
|
662
|
+
for (const line of lines) console.log(line);
|
|
663
|
+
} else {
|
|
664
|
+
for (const line of lines) console.error(line);
|
|
665
|
+
}
|
|
666
|
+
process.exit(result.ok ? 0 : 1);
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
// Optional XRPL anchor verbs — operator-run, off by default, NOT in the normal
|
|
670
|
+
// ingest/CI path. Like --verify-chain these are standalone audit/operations:
|
|
671
|
+
// no submission, no stdin, no provenance adapter. --anchor-compute and
|
|
672
|
+
// --anchor-verify are fully offline (never import xrpl); --anchor-post lazily
|
|
673
|
+
// loads the optional xrpl package and needs XRPL_SEED. Each handler returns
|
|
674
|
+
// { ok, exitCode, lines, event } and run.js owns the console + logStage + exit.
|
|
675
|
+
if (anchorComputeFlag || anchorPostFlag || anchorVerifyFlag) {
|
|
676
|
+
const correlation_id = synthCorrelationId();
|
|
677
|
+
let result;
|
|
678
|
+
if (anchorComputeFlag) {
|
|
679
|
+
result = handleAnchorCompute(repoRoot, {
|
|
680
|
+
mode: anchorMode,
|
|
681
|
+
...(anchorAlgo ? { algo: anchorAlgo } : {}),
|
|
682
|
+
...(anchorNetwork ? { network: anchorNetwork } : {}),
|
|
683
|
+
});
|
|
684
|
+
} else if (anchorPostFlag) {
|
|
685
|
+
result = await handleAnchorPost(repoRoot, {
|
|
686
|
+
mode: anchorMode,
|
|
687
|
+
...(anchorNetwork ? { network: anchorNetwork } : {}),
|
|
688
|
+
});
|
|
689
|
+
} else {
|
|
690
|
+
// --anchor-verify: optionally load a fetched tx JSON for the on-chain leg.
|
|
691
|
+
let tx;
|
|
692
|
+
if (anchorTxFile) {
|
|
693
|
+
const { readFileSync } = await import('node:fs');
|
|
694
|
+
try {
|
|
695
|
+
tx = JSON.parse(readFileSync(resolve(anchorTxFile), 'utf-8'));
|
|
696
|
+
} catch (err) {
|
|
697
|
+
emitCliErrorEvent({
|
|
698
|
+
failedStage: 'anchor_verify_read_tx',
|
|
699
|
+
correlationId: correlation_id,
|
|
700
|
+
err,
|
|
701
|
+
humanPrefix: 'could not read --anchor-tx file'
|
|
702
|
+
});
|
|
703
|
+
process.exit(2);
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
result = handleAnchorVerify(repoRoot, { tx, trustedAnchorAccounts: anchorTrustedAccounts });
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
// logStage strips any inner `stage:` field (the positional name wins), so
|
|
710
|
+
// spreading result.event — which carries its own `stage` — is safe.
|
|
711
|
+
logStage(result.event.stage, { correlation_id, ...result.event });
|
|
712
|
+
const sink = result.exitCode === 0 ? console.log : console.error;
|
|
713
|
+
for (const line of result.lines) sink(line);
|
|
714
|
+
process.exit(result.exitCode);
|
|
715
|
+
}
|
|
716
|
+
|
|
571
717
|
if (!submissionJson) {
|
|
572
718
|
// Read from stdin
|
|
573
719
|
const chunks = [];
|
|
@@ -629,32 +775,36 @@ if (isMain) {
|
|
|
629
775
|
console.error('WARNING: Using stub provenance (test/dev only). Records will NOT have real provenance verification.');
|
|
630
776
|
provenance = stubProvenance;
|
|
631
777
|
} else if (provenanceMode === 'github') {
|
|
632
|
-
|
|
633
|
-
|
|
778
|
+
// --provenance=github selects REAL provenance; the actual provider is taken
|
|
779
|
+
// from submission.source.provider, so a GitLab submission is confirmed via
|
|
780
|
+
// gitlabProvenance end-to-end (the adapter registry keys on the provider).
|
|
781
|
+
const resolved = resolveProviderProvenance(submission);
|
|
782
|
+
if (resolved.err) {
|
|
634
783
|
emitCliErrorEvent({
|
|
635
784
|
failedStage: 'cli_provenance_resolve',
|
|
636
785
|
correlationId: cliCorrelationId,
|
|
637
786
|
submissionId: submission && submission.run_id ? submission.run_id : null,
|
|
638
|
-
err:
|
|
787
|
+
err: resolved.err,
|
|
639
788
|
humanPrefix: 'provenance precondition unmet'
|
|
640
789
|
});
|
|
641
790
|
process.exit(2);
|
|
642
791
|
}
|
|
643
|
-
provenance =
|
|
792
|
+
provenance = resolved.provenance;
|
|
644
793
|
} else if (process.env.CI === 'true' || process.env.GITHUB_ACTIONS === 'true') {
|
|
645
|
-
// In CI without explicit flag: default to
|
|
646
|
-
|
|
647
|
-
|
|
794
|
+
// In CI without an explicit flag: default to real provenance, routed by the
|
|
795
|
+
// submission's source.provider (github | gitlab).
|
|
796
|
+
const resolved = resolveProviderProvenance(submission);
|
|
797
|
+
if (resolved.err) {
|
|
648
798
|
emitCliErrorEvent({
|
|
649
799
|
failedStage: 'cli_provenance_resolve',
|
|
650
800
|
correlationId: cliCorrelationId,
|
|
651
801
|
submissionId: submission && submission.run_id ? submission.run_id : null,
|
|
652
|
-
err:
|
|
802
|
+
err: resolved.err,
|
|
653
803
|
humanPrefix: 'provenance precondition unmet'
|
|
654
804
|
});
|
|
655
805
|
process.exit(2);
|
|
656
806
|
}
|
|
657
|
-
provenance =
|
|
807
|
+
provenance = resolved.provenance;
|
|
658
808
|
} else {
|
|
659
809
|
emitCliErrorEvent({
|
|
660
810
|
failedStage: 'cli_provenance_resolve',
|