@dogfood-lab/ingest 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,7 +8,8 @@
8
8
  *
9
9
  * Regenerated on every accepted/rejected write in Phase 1.
10
10
  *
11
- * Multi-file commit-group atomicity (W3-PIPE-002):
11
+ * Multi-file commit-group: crash/IO-failure RECOVERY-atomic, NOT reader-atomic
12
+ * (W3-PIPE-002):
12
13
  * The 3 indexes are written together via a two-phase commit pattern. Phase 1
13
14
  * stages all 3 files to temp paths AND records them in a journal file. Phase 2
14
15
  * renames each temp into its final location, then deletes the journal. If the
@@ -17,10 +18,27 @@
17
18
  * is idempotent (it scans records/ end-to-end), so re-running is the correct
18
19
  * recovery action.
19
20
  *
20
- * Pattern reference: choke-point fix (Pattern #4) for multi-file atomicity.
21
+ * IMPORTANT — what "atomicity" means here. The guarantee is RECOVERY-atomic,
22
+ * not READER-atomic. Phase 2 promotes the temps with a per-leg `renameSync`
23
+ * (each rename is individually atomic), but the GROUP is not promoted under a
24
+ * single atomic operation. During the promote window — and during the heal
25
+ * window after a mid-promote IO failure (ENOSPC/EACCES after the first
26
+ * final is renamed but a later one is not) — a concurrent reader CAN observe
27
+ * the index group in a mutually-inconsistent intermediate state (e.g. an
28
+ * already-promoted latest-by-repo.json against a not-yet-promoted failing.json).
29
+ * The catch on a promote failure does NOT roll back already-promoted finals;
30
+ * it preserves the journal and emits a structured error event so an operator
31
+ * can force an immediate rebuild before the next scheduled run heals it. The
32
+ * design is sound because the only writer (the ingest pipeline) serializes
33
+ * rebuilds and `rebuildIndexes` is synchronous — there is no in-flight reader
34
+ * that races a writer mid-promote within a single process. If you ever need
35
+ * true reader-atomicity (a reader that NEVER sees a torn group), this design
36
+ * must change (e.g. swap a single directory symlink, or version the index dir).
37
+ *
38
+ * Pattern reference: choke-point fix (Pattern #4) for multi-file recovery.
21
39
  * Single-file `atomicWriteFileSync` (lib/atomic-write.js) handles each leg;
22
- * the journal handles the cross-file boundary. The single-file helper is
23
- * the same one Class #6 helper-adoption-sweep enforces as canonical for
40
+ * the journal handles the cross-file recovery boundary. The single-file helper
41
+ * is the same one Class #6 helper-adoption-sweep enforces as canonical for
24
42
  * temp+rename writes under `packages/ingest/`.
25
43
  */
26
44
 
@@ -98,6 +116,72 @@ export function findJsonFiles(dir) {
98
116
  return results;
99
117
  }
100
118
 
119
+ /**
120
+ * Probe whether `dir` exists but is unreadable at its OWN level (EACCES /
121
+ * Windows lock / ENOTDIR) — as distinct from a deep leaf failing mid-walk.
122
+ *
123
+ * ingest-B-002: `findJsonFiles` deliberately degrades an unreadable subtree to
124
+ * "those records are missing" and returns `[]`. That is correct for a single
125
+ * locked LEAF, but catastrophic for the records/ ROOT: a transiently-locked
126
+ * root makes the WHOLE corpus invisible, and an unguarded rebuild would then
127
+ * overwrite every index with empty content. This probe lets `rebuildIndexes`
128
+ * tell the two apart so it can REFUSE to clobber good indexes when the root
129
+ * itself is the thing that failed. A non-existent dir is NOT unreadable — that
130
+ * is the legitimate empty-corpus case, which must still rebuild empty indexes.
131
+ *
132
+ * @param {string} dir
133
+ * @returns {{ unreadable: boolean, code: string|null, error: string|null }}
134
+ */
135
+ function probeDirReadable(dir) {
136
+ if (!existsSync(dir)) return { unreadable: false, code: null, error: null };
137
+ try {
138
+ readdirSync(dir);
139
+ return { unreadable: false, code: null, error: null };
140
+ } catch (err) {
141
+ return {
142
+ unreadable: true,
143
+ code: err && err.code ? err.code : 'readdir_failed',
144
+ error: err && err.message ? err.message : String(err),
145
+ };
146
+ }
147
+ }
148
+
149
+ /**
150
+ * Read the prior committed latest-by-repo.json so a rebuild can tell whether
151
+ * the index it is about to overwrite currently has content. Used by the
152
+ * ingest-B-002 refuse-to-overwrite guard: an empty scan is only suspicious if
153
+ * the prior index was non-empty. A missing or unparseable prior index counts
154
+ * as "no prior content" (the legitimate first-run / empty-corpus case).
155
+ *
156
+ * @param {string} latestPath
157
+ * @returns {boolean} true if the prior index existed and held at least one repo
158
+ */
159
+ function priorIndexHasContent(latestPath) {
160
+ if (!existsSync(latestPath)) return false;
161
+ try {
162
+ const prior = JSON.parse(readFileSync(latestPath, 'utf-8'));
163
+ return prior && typeof prior === 'object' && Object.keys(prior).length > 0;
164
+ } catch {
165
+ // Unparseable prior index — treat as no usable content so a corrupt index
166
+ // never wedges the rebuild into a permanent refuse state.
167
+ return false;
168
+ }
169
+ }
170
+
171
+ /**
172
+ * Whether the freshly-built latest-by-repo map has no repos. Used by the
173
+ * ingest-B-002 refuse-to-overwrite guard to recognise an empty scan. A scan
174
+ * can be empty because there are genuinely no accepted records (legitimate)
175
+ * or because the corpus was invisible (a transiently-locked records tree) —
176
+ * the guard combines this with `priorIndexHasContent` to tell them apart.
177
+ *
178
+ * @param {object} latestByRepo
179
+ * @returns {boolean}
180
+ */
181
+ function latestByRepoIsEmpty(latestByRepo) {
182
+ return !latestByRepo || Object.keys(latestByRepo).length === 0;
183
+ }
184
+
101
185
  /**
102
186
  * Load and parse a record file.
103
187
  *
@@ -131,8 +215,20 @@ export function rebuildIndexes(repoRoot, options = {}) {
131
215
  const indexDir = join(repoRoot, 'indexes');
132
216
  mkdirSync(indexDir, { recursive: true });
133
217
 
134
- // Collect all records (accepted + rejected)
218
+ // ingest-B-002: capture two facts BEFORE the scan so we can refuse to clobber
219
+ // good indexes with empty ones when the corpus is invisible rather than empty.
220
+ // 1. Is the records/ ROOT itself unreadable (vs a deep leaf, vs absent)?
221
+ // A locked root makes the WHOLE corpus invisible — findJsonFiles would
222
+ // return [] after one low `dir_unreadable` warn, and an unguarded
223
+ // commit-group would then overwrite every index with {}.
224
+ // 2. Did the PRIOR latest-by-repo.json have content? An empty scan is only
225
+ // suspicious if there was something to lose; a legitimately empty
226
+ // first-run corpus must still write empty indexes.
135
227
  const recordsDir = join(repoRoot, 'records');
228
+ const latestPath = join(indexDir, 'latest-by-repo.json');
229
+ const rootProbe = probeDirReadable(recordsDir);
230
+ const hadPriorIndex = priorIndexHasContent(latestPath);
231
+
136
232
  const acceptedFiles = findJsonFiles(recordsDir)
137
233
  .filter(f => {
138
234
  const rel = relative(recordsDir, f);
@@ -268,9 +364,49 @@ export function rebuildIndexes(repoRoot, options = {}) {
268
364
  }
269
365
  }
270
366
 
367
+ // ingest-B-002: REFUSE to overwrite good indexes with empty ones when the
368
+ // corpus was invisible rather than genuinely empty. Two refuse conditions:
369
+ // - records_root_unreadable: the records/ ROOT itself failed to read
370
+ // (EACCES / Windows lock / ENOTDIR). The entire corpus is invisible —
371
+ // committing now would wipe every index. This is distinct from a single
372
+ // locked leaf, which findJsonFiles already degrades to a partial scan.
373
+ // - empty_scan_with_prior_index: the root read fine but the scan found
374
+ // zero accepted records while the prior latest-by-repo had content. The
375
+ // records likely vanished transiently; clobbering loses the portfolio.
376
+ // A legitimately empty corpus (no accepted records AND no prior content) is
377
+ // NOT refused — it must still write empty indexes (first-run case). We skip
378
+ // the commit-group and emit a structured, greppable event so the operator
379
+ // sees the refusal loudly instead of a silently-emptied portfolio.
380
+ const noAcceptedScanned = latestByRepoIsEmpty(latestByRepo);
381
+ if (rootProbe.unreadable || (noAcceptedScanned && hadPriorIndex)) {
382
+ const reason = rootProbe.unreadable
383
+ ? 'records_root_unreadable'
384
+ : 'empty_scan_with_prior_index';
385
+ // A root IO failure is an operator-actionable error (the corpus is gone);
386
+ // an empty scan over a readable root is a warn (recoverable next run).
387
+ logStage(rootProbe.unreadable ? 'error' : 'warn', {
388
+ kind: 'index_rebuild_skipped',
389
+ reason,
390
+ records_dir: recordsDir,
391
+ accepted_scanned: acceptedFiles.length,
392
+ prior_index_non_empty: hadPriorIndex,
393
+ root_error_code: rootProbe.code,
394
+ error: rootProbe.error,
395
+ });
396
+ return {
397
+ latestByRepo,
398
+ failing,
399
+ stale,
400
+ accepted: acceptedFiles.length,
401
+ rejected: rejectedFiles.length,
402
+ corrupted,
403
+ skipped,
404
+ skippedCommit: reason,
405
+ };
406
+ }
407
+
271
408
  // Write indexes via commit-group two-phase commit. See module header
272
409
  // for the full design rationale.
273
- const latestPath = join(indexDir, 'latest-by-repo.json');
274
410
  const failingPath = join(indexDir, 'failing.json');
275
411
  const stalePath = join(indexDir, 'stale.json');
276
412
 
@@ -307,15 +443,23 @@ export function rebuildIndexes(repoRoot, options = {}) {
307
443
  * AND records them in a journal first; then renames them in caller-given
308
444
  * order. The journal is deleted only after every rename succeeds.
309
445
  *
310
- * Crash semantics:
311
- * - Crash during STAGE phase: every staged temp is unlinked in the catch
446
+ * Crash / IO-failure semantics (RECOVERY-atomic, not reader-atomic):
447
+ * - Failure during STAGE phase: every staged temp is unlinked in the catch
312
448
  * block; the journal (if written) is unlinked too. No partial visible
313
- * state.
314
- * - Crash during PROMOTE phase: any successfully-renamed file is at its
315
- * final path; remaining temps are still next to their finals. The
316
- * journal still exists. Next run's `cleanupCrashedJournals` deletes
317
- * residual temps and the journal; the next normal `rebuildIndexes`
318
- * call rewrites all 3 indexes from scratch (idempotent).
449
+ * state — no final was touched.
450
+ * - Failure during PROMOTE phase: any successfully-renamed file is at its
451
+ * final path with its NEW content; remaining temps are still next to
452
+ * their (still-OLD) finals. The group is therefore mutually inconsistent
453
+ * until healed — a reader in this window sees a torn group. We do NOT
454
+ * roll back the already-promoted finals (their prior content was already
455
+ * overwritten by the atomic rename — there is nothing to roll back to
456
+ * without re-reading the journal). Instead we emit a structured
457
+ * `logStage('error', { kind: 'commit_group_partial_promote', ... })`
458
+ * naming which finals were promoted vs left stale so an operator can
459
+ * force an immediate rebuild, and we preserve the journal. Next run's
460
+ * `cleanupCrashedJournals` deletes residual temps and the journal; the
461
+ * next normal `rebuildIndexes` call rewrites all 3 indexes from scratch
462
+ * (idempotent), which is what heals the torn group.
319
463
  *
320
464
  * Why journal-then-rename rather than journal-only: the rename phase needs
321
465
  * to be the visible commit point. A journal-only design would require
@@ -339,8 +483,12 @@ function commitGroupRename(indexDir, entries) {
339
483
 
340
484
  // Write journal AFTER staging so it never points at a non-existent temp.
341
485
  // Atomic write of the journal itself: writeFileSync directly is fine here
342
- // because the journal is process-private (the pid suffix guarantees no
343
- // collision with concurrent rebuilds).
486
+ // because the journal is process-private — the pid + random suffix make
487
+ // the filename collision-free, and `cleanupCrashedJournals` is pid-aware
488
+ // (it skips journals whose pid is a still-live process), so a future
489
+ // concurrent rebuild's in-flight journal is never reaped out from under
490
+ // it. The temp `entries` it lists are equally collision-free (each carries
491
+ // its own random suffix from `stageWriteFileSync`).
344
492
  writeFileSync(
345
493
  journalPath,
346
494
  JSON.stringify({
@@ -374,6 +522,24 @@ function commitGroupRename(indexDir, entries) {
374
522
  // (their previous content is already overwritten — the rename was
375
523
  // atomic at each individual leg, just not as a group). The next run
376
524
  // is idempotent and will rewrite all three from scratch.
525
+ //
526
+ // ingest-A-001: the group is now reader-inconsistent (promoted finals
527
+ // carry new content; stale finals carry old content). Name which finals
528
+ // are which in a structured error event so an operator can force an
529
+ // immediate rebuild rather than wait for the next scheduled run to heal
530
+ // the torn group.
531
+ const promoted = stagedTmps.slice(0, promotedCount).map((e) => e.finalPath);
532
+ const stale = stagedTmps.slice(promotedCount).map((e) => e.finalPath);
533
+ logStage('error', {
534
+ kind: 'commit_group_partial_promote',
535
+ reason: err && err.code ? err.code : 'promote_failed',
536
+ promoted_count: promotedCount,
537
+ total: stagedTmps.length,
538
+ promoted_finals: promoted,
539
+ stale_finals: stale,
540
+ journal: journalPath,
541
+ error: err && err.message ? err.message : String(err),
542
+ });
377
543
  throw new Error(
378
544
  `commitGroupRename: promote failed after ${promotedCount}/${stagedTmps.length} files; ` +
379
545
  `journal preserved at ${journalPath} for next-run cleanup. Original error: ${err.message}`
@@ -387,11 +553,42 @@ function commitGroupRename(indexDir, entries) {
387
553
  try { unlinkSync(journalPath); } catch { /* will be cleaned next run */ }
388
554
  }
389
555
 
556
+ /**
557
+ * Probe whether a pid is still a live process. `process.kill(pid, 0)` sends
558
+ * no signal — it only performs the permission/existence check, throwing
559
+ * ESRCH when the pid is dead. An EPERM means the process exists but is owned
560
+ * by another user; that still counts as "live" for our purpose (do not reap
561
+ * its journal). Any other error (or a non-integer pid) is treated as "not
562
+ * provably live" so a malformed journal never blocks its own cleanup.
563
+ *
564
+ * @param {unknown} pid
565
+ * @returns {boolean}
566
+ */
567
+ function isProcessAlive(pid) {
568
+ if (!Number.isInteger(pid) || pid <= 0) return false;
569
+ try {
570
+ process.kill(pid, 0);
571
+ return true;
572
+ } catch (err) {
573
+ return err && err.code === 'EPERM';
574
+ }
575
+ }
576
+
390
577
  /**
391
578
  * Find and clean up any in-progress journals from previous runs. Each journal
392
579
  * lists the temp paths that were staged; we unlink any that still exist
393
580
  * (they are residue from a crashed run) and delete the journal.
394
581
  *
582
+ * ingest-A-002: cleanup is PID-AWARE. A journal whose `pid` is a still-live
583
+ * process is the in-flight recovery state of a concurrent rebuild — reaping
584
+ * it would delete that run's temps and journal mid-flight. Today the only
585
+ * writer serializes rebuilds and `rebuildIndexes` is synchronous, so no live
586
+ * sibling journal exists at Phase-0 cleanup time; this guard makes the design
587
+ * correct (not merely safe-by-serialization) so a future maintainer who adds
588
+ * concurrency does not silently corrupt a peer. A dead pid, a missing/
589
+ * malformed pid, or an unreadable journal is still reaped — that is the
590
+ * crashed-run residue this function exists to clear.
591
+ *
395
592
  * Idempotent: on a clean filesystem it's a no-op; on a crashed-mid-promote
396
593
  * filesystem it cleans the slate so the upcoming `commitGroupRename` can
397
594
  * stage fresh temps without colliding.
@@ -412,6 +609,11 @@ function cleanupCrashedJournals(indexDir) {
412
609
  // referenced will linger but they're harmless (they have a unique
413
610
  // suffix that won't be re-used).
414
611
  }
612
+ // Skip a journal owned by a still-live process — it belongs to a
613
+ // concurrent rebuild's in-flight recovery state, not crashed residue.
614
+ if (parsed && isProcessAlive(parsed.pid) && parsed.pid !== process.pid) {
615
+ continue;
616
+ }
415
617
  if (parsed && Array.isArray(parsed.entries)) {
416
618
  for (const e of parsed.entries) {
417
619
  if (e && typeof e.tmpPath === 'string') {
package/run.js CHANGED
@@ -22,14 +22,44 @@ import { fileURLToPath } from 'node:url';
22
22
  import { randomBytes } from 'node:crypto';
23
23
 
24
24
  import { verify } from '@dogfood-lab/verify';
25
- import { stubProvenance, githubProvenance } from '@dogfood-lab/verify/validators/provenance.js';
25
+ import { stubProvenance, provenanceForProvider } from '@dogfood-lab/verify/validators/provenance.js';
26
26
  import { logStage as sharedLogStage } from '@dogfood-lab/dogfood-swarm/lib/log-stage.js';
27
27
  import { loadGlobalPolicy, loadRepoPolicy, loadScenarios } from './load-context.js';
28
28
  import { isDuplicate, writeRecord, computeRecordPath } from './persist.js';
29
29
  import { rebuildIndexes } from './rebuild-indexes.js';
30
+ import { verifyChain, formatChainResult } from './verify-chain.js';
31
+ import { handleAnchorCompute, handleAnchorPost, handleAnchorVerify } from './anchor/cli.js';
30
32
 
31
33
  const __dirname = dirname(fileURLToPath(import.meta.url));
32
34
 
35
+ /**
36
+ * Resolve the REAL provenance adapter for a submission, routed by
37
+ * `submission.source.provider` (github | gitlab), sourcing the provider's token
38
+ * from the environment. Returns `{ provenance }` on success or `{ err }` (a
39
+ * structured, operator-legible Error the caller surfaces via emitCliErrorEvent
40
+ * + exit 2). The adapter registry (`provenanceForProvider`) is the single
41
+ * provider-keyed seam; a provider in the schema enum without a registered
42
+ * adapter fails here loudly rather than silently skipping verification.
43
+ *
44
+ * @param {object} submission
45
+ * @returns {{ provenance: object } | { err: Error }}
46
+ */
47
+ function resolveProviderProvenance(submission) {
48
+ const provider = (submission && submission.source && submission.source.provider) || 'github';
49
+ const factory = provenanceForProvider(provider);
50
+ if (!factory) {
51
+ return { err: new Error(`unknown provenance provider '${provider}' — no adapter registered (supported: github, gitlab).`) };
52
+ }
53
+ const token = provider === 'gitlab'
54
+ ? (process.env.GITLAB_TOKEN || process.env.CI_JOB_TOKEN)
55
+ : (process.env.GITHUB_TOKEN || process.env.GH_TOKEN);
56
+ if (!token) {
57
+ const need = provider === 'gitlab' ? 'GITLAB_TOKEN or CI_JOB_TOKEN' : 'GITHUB_TOKEN or GH_TOKEN';
58
+ return { err: new Error(`real provenance for provider '${provider}' requires ${need} in the environment.`) };
59
+ }
60
+ return { provenance: factory(token) };
61
+ }
62
+
33
63
  /**
34
64
  * SEED-1 (d3-ingest-003) — posixify a path-shaped value at the operator/log
35
65
  * SERIALIZATION boundary. `computeRecordPath`/`writeRecord` return OS-native
@@ -513,6 +543,18 @@ if (isMain) {
513
543
  let submissionJson;
514
544
  let provenanceMode = null;
515
545
  let verifyOnlyFlag = false;
546
+ let verifyChainFlag = false;
547
+ // Anchor verbs (optional, off-by-default, operator-run). --anchor-compute and
548
+ // --anchor-verify are fully offline (never import xrpl); --anchor-post lazily
549
+ // loads the optional xrpl package and needs XRPL_SEED.
550
+ let anchorComputeFlag = false;
551
+ let anchorPostFlag = false;
552
+ let anchorVerifyFlag = false;
553
+ let anchorMode = 'since-last';
554
+ let anchorAlgo = null;
555
+ let anchorNetwork = null;
556
+ let anchorTxFile = null;
557
+ let anchorTrustedAccounts = [];
516
558
  const positionalArgs = [];
517
559
 
518
560
  for (let i = 0; i < args.length; i++) {
@@ -563,11 +605,115 @@ if (isMain) {
563
605
  // F-252714-058: dry-run the pipeline without writing or rebuilding
564
606
  // indexes. CI / operators preview what WOULD have been persisted.
565
607
  verifyOnlyFlag = true;
608
+ } else if (arg === '--verify-chain') {
609
+ // Integrity chain v1: verify the append-only tamper-evident ledger at
610
+ // indexes/integrity/chain.jsonl, fully offline. No submission, no stdin,
611
+ // no provenance — a standalone audit command.
612
+ verifyChainFlag = true;
613
+ } else if (arg === '--anchor-compute') {
614
+ // Optional XRPL anchor: compute + write the next anchor manifest. Offline.
615
+ anchorComputeFlag = true;
616
+ } else if (arg === '--anchor-post') {
617
+ // Optional XRPL anchor: compute if needed + post to XRPL. Needs the
618
+ // optional xrpl package (lazily loaded) and XRPL_SEED.
619
+ anchorPostFlag = true;
620
+ } else if (arg === '--anchor-verify') {
621
+ // Optional XRPL anchor: verify local manifests + run the truncation check.
622
+ // Offline reports honest NOT-verified for the on-chain leg.
623
+ anchorVerifyFlag = true;
624
+ } else if (arg === '--anchor-all') {
625
+ // Genesis snapshot mode for compute/post (covers the whole chain).
626
+ anchorMode = 'all';
627
+ } else if (arg === '--anchor-algo' && hasValue) {
628
+ anchorAlgo = takeValue();
629
+ } else if (arg === '--anchor-network' && hasValue) {
630
+ anchorNetwork = takeValue();
631
+ } else if (arg === '--anchor-tx' && hasValue) {
632
+ // Path to a JSON file containing a fetched XRPL tx (with Memos) for the
633
+ // on-chain leg of --anchor-verify. Offline-honest: omit it to run the
634
+ // truncation check only.
635
+ anchorTxFile = takeValue();
636
+ } else if (arg === '--anchor-trusted' && hasValue) {
637
+ // Comma-separated trusted anchor accounts (UNIONed with the bundled list).
638
+ anchorTrustedAccounts = takeValue().split(',').map((s) => s.trim()).filter(Boolean);
566
639
  } else {
567
640
  positionalArgs.push(args[i]);
568
641
  }
569
642
  }
570
643
 
644
+ // --verify-chain is a standalone, side-effect-free audit: it reads only the
645
+ // ledger + the record files it references, takes no submission, reads no
646
+ // stdin, and needs no provenance adapter. Handle it BEFORE the stdin read and
647
+ // provenance resolution so `node run.js --verify-chain` does not block on
648
+ // stdin or demand a --provenance flag. Exit 0 when the chain verifies, 1 on
649
+ // the first break (operator-legible output, no raw stack traces).
650
+ if (verifyChainFlag) {
651
+ const result = verifyChain(repoRoot);
652
+ logStage(result.ok ? 'verify_chain_complete' : 'error', {
653
+ correlation_id: synthCorrelationId(),
654
+ ...(result.ok ? {} : { failed_stage: 'verify_chain' }),
655
+ verified: result.count,
656
+ head_digest: result.head_digest,
657
+ chain_ok: result.ok,
658
+ ...(result.break ? { break_seq: result.break.seq, break_reason: result.break.reason } : {})
659
+ });
660
+ const lines = formatChainResult(result);
661
+ if (result.ok) {
662
+ for (const line of lines) console.log(line);
663
+ } else {
664
+ for (const line of lines) console.error(line);
665
+ }
666
+ process.exit(result.ok ? 0 : 1);
667
+ }
668
+
669
+ // Optional XRPL anchor verbs — operator-run, off by default, NOT in the normal
670
+ // ingest/CI path. Like --verify-chain these are standalone audit/operations:
671
+ // no submission, no stdin, no provenance adapter. --anchor-compute and
672
+ // --anchor-verify are fully offline (never import xrpl); --anchor-post lazily
673
+ // loads the optional xrpl package and needs XRPL_SEED. Each handler returns
674
+ // { ok, exitCode, lines, event } and run.js owns the console + logStage + exit.
675
+ if (anchorComputeFlag || anchorPostFlag || anchorVerifyFlag) {
676
+ const correlation_id = synthCorrelationId();
677
+ let result;
678
+ if (anchorComputeFlag) {
679
+ result = handleAnchorCompute(repoRoot, {
680
+ mode: anchorMode,
681
+ ...(anchorAlgo ? { algo: anchorAlgo } : {}),
682
+ ...(anchorNetwork ? { network: anchorNetwork } : {}),
683
+ });
684
+ } else if (anchorPostFlag) {
685
+ result = await handleAnchorPost(repoRoot, {
686
+ mode: anchorMode,
687
+ ...(anchorNetwork ? { network: anchorNetwork } : {}),
688
+ });
689
+ } else {
690
+ // --anchor-verify: optionally load a fetched tx JSON for the on-chain leg.
691
+ let tx;
692
+ if (anchorTxFile) {
693
+ const { readFileSync } = await import('node:fs');
694
+ try {
695
+ tx = JSON.parse(readFileSync(resolve(anchorTxFile), 'utf-8'));
696
+ } catch (err) {
697
+ emitCliErrorEvent({
698
+ failedStage: 'anchor_verify_read_tx',
699
+ correlationId: correlation_id,
700
+ err,
701
+ humanPrefix: 'could not read --anchor-tx file'
702
+ });
703
+ process.exit(2);
704
+ }
705
+ }
706
+ result = handleAnchorVerify(repoRoot, { tx, trustedAnchorAccounts: anchorTrustedAccounts });
707
+ }
708
+
709
+ // logStage strips any inner `stage:` field (the positional name wins), so
710
+ // spreading result.event — which carries its own `stage` — is safe.
711
+ logStage(result.event.stage, { correlation_id, ...result.event });
712
+ const sink = result.exitCode === 0 ? console.log : console.error;
713
+ for (const line of result.lines) sink(line);
714
+ process.exit(result.exitCode);
715
+ }
716
+
571
717
  if (!submissionJson) {
572
718
  // Read from stdin
573
719
  const chunks = [];
@@ -629,32 +775,36 @@ if (isMain) {
629
775
  console.error('WARNING: Using stub provenance (test/dev only). Records will NOT have real provenance verification.');
630
776
  provenance = stubProvenance;
631
777
  } else if (provenanceMode === 'github') {
632
- const token = process.env.GITHUB_TOKEN || process.env.GH_TOKEN;
633
- if (!token) {
778
+ // --provenance=github selects REAL provenance; the actual provider is taken
779
+ // from submission.source.provider, so a GitLab submission is confirmed via
780
+ // gitlabProvenance end-to-end (the adapter registry keys on the provider).
781
+ const resolved = resolveProviderProvenance(submission);
782
+ if (resolved.err) {
634
783
  emitCliErrorEvent({
635
784
  failedStage: 'cli_provenance_resolve',
636
785
  correlationId: cliCorrelationId,
637
786
  submissionId: submission && submission.run_id ? submission.run_id : null,
638
- err: new Error('--provenance=github requires GITHUB_TOKEN or GH_TOKEN environment variable.'),
787
+ err: resolved.err,
639
788
  humanPrefix: 'provenance precondition unmet'
640
789
  });
641
790
  process.exit(2);
642
791
  }
643
- provenance = githubProvenance(token);
792
+ provenance = resolved.provenance;
644
793
  } else if (process.env.CI === 'true' || process.env.GITHUB_ACTIONS === 'true') {
645
- // In CI without explicit flag: default to github provenance, fail if no token
646
- const token = process.env.GITHUB_TOKEN || process.env.GH_TOKEN;
647
- if (!token) {
794
+ // In CI without an explicit flag: default to real provenance, routed by the
795
+ // submission's source.provider (github | gitlab).
796
+ const resolved = resolveProviderProvenance(submission);
797
+ if (resolved.err) {
648
798
  emitCliErrorEvent({
649
799
  failedStage: 'cli_provenance_resolve',
650
800
  correlationId: cliCorrelationId,
651
801
  submissionId: submission && submission.run_id ? submission.run_id : null,
652
- err: new Error('Running in CI without --provenance flag and no GITHUB_TOKEN. Cannot verify provenance.'),
802
+ err: resolved.err,
653
803
  humanPrefix: 'provenance precondition unmet'
654
804
  });
655
805
  process.exit(2);
656
806
  }
657
- provenance = githubProvenance(token);
807
+ provenance = resolved.provenance;
658
808
  } else {
659
809
  emitCliErrorEvent({
660
810
  failedStage: 'cli_provenance_resolve',