gitnexus 1.6.10-rc.80 → 1.6.10-rc.82

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,7 +14,7 @@ import v8 from 'v8';
14
14
  import cliProgress from 'cli-progress';
15
15
  import { isLbugReady, LbugWipeError } from '../core/lbug/lbug-adapter.js';
16
16
  import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js';
17
- import { getOsPageSize, isLbugCheckpointIoError, isLbugPageSizeFrameError, isPageSizeAwareLadybug, isWalCorruptionError, parseWalCheckpointThreshold, WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js';
17
+ import { getOsPageSize, isLbugCheckpointIoError, isLbugCheckpointBusyError, isLbugPageSizeFrameError, isPageSizeAwareLadybug, isWalCorruptionError, parseWalCheckpointThreshold, WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js';
18
18
  import { getStoragePaths, getGlobalRegistryPath, RegistryNameCollisionError, AnalysisNotFinalizedError, assertAnalysisFinalized, } from '../storage/repo-manager.js';
19
19
  import { getGitRoot, hasGitDir, getDefaultBranch } from '../storage/git.js';
20
20
  import { loadAnalyzeConfig, mergeAnalyzeOptions, resolveDefaultBranch, validateBranchName, GitNexusRcError, } from './analyze-config.js';
@@ -1236,7 +1236,15 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
1236
1236
  return;
1237
1237
  }
1238
1238
  if (isLbugCheckpointIoError(err)) {
1239
+ // #2599: when the checkpoint IO error also looks busy/locked, another
1240
+ // handle holds the store open — name that actionable cause alongside the
1241
+ // threshold hint (the original error is preserved so the hint still fires).
1242
+ const heldOpen = isLbugCheckpointBusyError(err)
1243
+ ? ` Another process may hold the store open (a running \`gitnexus mcp\` server, or a\n` +
1244
+ ` stale reader) — close other GitNexus processes on this repo, then retry.\n`
1245
+ : '';
1239
1246
  cliError(` LadybugDB failed while rotating/removing WAL checkpoint files.\n` +
1247
+ heldOpen +
1240
1248
  ` This can happen when auto-checkpoint runs at the default threshold (~16MB).\n` +
1241
1249
  ` Retry with a larger checkpoint threshold to reduce checkpoint frequency:\n` +
1242
1250
  ` gitnexus analyze --wal-checkpoint-threshold ${RECOMMENDED_WAL_CHECKPOINT_THRESHOLD}\n` +
@@ -829,6 +829,8 @@ const LBUG_OPEN_RETRY_PATTERNS = [
829
829
  'could not set lock',
830
830
  'lock held by another process',
831
831
  ];
832
+ // Cross-repo bridge RO open retry. Catalogued as entry 5 of the lbug-config
833
+ // retry-budget registry; caps back-off so total wait ~3s.
832
834
  const LBUG_OPEN_RETRY_ATTEMPTS = 10;
833
835
  const LBUG_OPEN_RETRY_BASE_MS = 100;
834
836
  /** Cap individual back-off delays so the total wait is bounded (~3s). */
@@ -128,6 +128,15 @@ export interface LbugConnectionHandle {
128
128
  * import directly from this module — no re-export to keep in sync.
129
129
  */
130
130
  export declare const isDbBusyError: (err: unknown) => boolean;
131
+ /**
132
+ * True when a WAL-checkpoint IO error ALSO carries a busy/lock signal — the
133
+ * rotation failed because another handle (a `gitnexus mcp` server, or this
134
+ * process's own reader) holds the store's WAL open, rather than a permanent
135
+ * disk error. Reuses `isDbBusyError`'s already-tested keyword set instead of a
136
+ * fresh regex, so an unmatched message degrades to "IO error" rather than
137
+ * silently claiming a held-open cause. (#2599)
138
+ */
139
+ export declare const isLbugCheckpointBusyError: (err: unknown) => boolean;
131
140
  /** See {@link classifyDeleteAllError}. */
132
141
  export type DeleteAllErrorClass = 'benign-missing-table' | 'rethrow';
133
142
  /**
@@ -582,6 +582,27 @@ export const isDbBusyError = (err) => {
582
582
  msg.includes('already in use') ||
583
583
  msg.includes('only one write transaction at a time'));
584
584
  };
585
+ /**
586
+ * True when a WAL-checkpoint IO error ALSO carries a busy/lock signal — the
587
+ * rotation failed because another handle (a `gitnexus mcp` server, or this
588
+ * process's own reader) holds the store's WAL open, rather than a permanent
589
+ * disk error. Reuses `isDbBusyError`'s already-tested keyword set instead of a
590
+ * fresh regex, so an unmatched message degrades to "IO error" rather than
591
+ * silently claiming a held-open cause. (#2599)
592
+ */
593
+ export const isLbugCheckpointBusyError = (err) => {
594
+ if (!isLbugCheckpointIoError(err))
595
+ return false;
596
+ // Anchor to real held-open wording rather than isDbBusyError's broad
597
+ // `.includes('lock')`, which matches the DB PATH embedded in the checkpoint
598
+ // error message (e.g. a repo under `blockchain-app`) and would misclassify a
599
+ // pure disk fault as held-open (#2614 LOW).
600
+ const msg = (err instanceof Error ? err.message : String(err)).toLowerCase();
601
+ return (msg.includes('could not set lock') ||
602
+ msg.includes('lock is held') ||
603
+ msg.includes('being used by another process') ||
604
+ msg.includes('is busy'));
605
+ };
585
606
  /**
586
607
  * Classify an error thrown while clearing all relationships of one type
587
608
  * before an incremental re-write (`deleteAllRelationshipsOfType` in
@@ -652,6 +673,18 @@ const OPEN_LOCK_RETRY_DELAY_MS = 100;
652
673
  export const HANDLE_RELEASE_PROBE_ATTEMPTS = 5;
653
674
  export const HANDLE_RELEASE_PROBE_DELAY_MS = 50;
654
675
  const HANDLE_RELEASE_LOCK_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']);
676
+ // Retry-budget registry, part 2 (retry-budget consolidation): the remaining
677
+ // open-time lock retries live next to their call sites but are catalogued here
678
+ // so all lbug retry budgets surface in one grep. They retry the same lock class
679
+ // as 1–3 ("Could not set lock" while a writer rebuilds the index):
680
+ // 4. LOCK_RETRY_ATTEMPTS / LOCK_RETRY_DELAY_MS (pool-adapter.ts)
681
+ // → read pool's read-only open while `gitnexus analyze` is writing
682
+ // (3 attempts, linear 2s·n back-off ≈ 6s total)
683
+ // 5. LBUG_OPEN_RETRY_ATTEMPTS / _BASE_MS / _MAX_MS (group/bridge-db.ts)
684
+ // → cross-repo bridge RO open race (10 attempts, linear 100ms·n capped
685
+ // at 500ms ≈ 3.5s total)
686
+ // Kept in-file (not moved here) so explicit `lbug-config` test mocks don't have
687
+ // to enumerate them; change a budget in its call site and update this catalogue.
655
688
  /**
656
689
  * Test-fixture directory prefixes recognized by `isTestFixturePath`.
657
690
  *
@@ -15,6 +15,21 @@
15
15
  * from the same Database is the officially supported concurrency pattern.
16
16
  */
17
17
  import lbug from '@ladybugdb/core';
18
+ /** Filesystem identity used to detect an index rebuilt/mutated under a live
19
+ * read pool. `ino` catches a full-rebuild unlink+recreate or an atomic-rename
20
+ * swap; `mtimeMs`+`size` catch an in-place incremental writeback. */
21
+ interface DbIdentity {
22
+ ino: number;
23
+ mtimeMs: number;
24
+ size: number;
25
+ }
26
+ export declare function statDbIdentity(dbPath: string): Promise<DbIdentity | null>;
27
+ /** True only when both identities are known AND differ. A stat failure
28
+ * (ENOENT during the brief unlink window of a full rebuild) yields false, so
29
+ * the reader keeps serving its still-valid open inode until the NEW file
30
+ * appears with a different identity — avoiding a churn into a failed reopen
31
+ * mid-rebuild. */
32
+ export declare function dbIdentityChanged(prev: DbIdentity | null, next: DbIdentity | null): boolean;
18
33
  /**
19
34
  * Listeners notified when a pool entry is torn down (LRU eviction, idle
20
35
  * timeout, explicit close). Used by upper layers (e.g. the BM25 search
@@ -87,7 +102,14 @@ export declare function restoreStdout(): void;
87
102
  * Concurrent calls for the same repoId are deduplicated — the second caller
88
103
  * awaits the first's in-progress init rather than starting a redundant one.
89
104
  */
90
- export declare const initLbug: (repoId: string, dbPath: string) => Promise<void>;
105
+ /**
106
+ * Returns `true` when this call (re)opened a fresh handle onto the current
107
+ * on-disk file, `false` when it reused/served the existing handle (unchanged,
108
+ * or changed-but-a-query-is-in-flight). Callers that gate their own freshness
109
+ * bookkeeping on "did the pool actually roll over" (LocalBackend) use the
110
+ * return value; callers that only need the pool ready can ignore it.
111
+ */
112
+ export declare const initLbug: (repoId: string, dbPath: string) => Promise<boolean>;
91
113
  /**
92
114
  * Initialize a pool entry from a pre-existing Database object.
93
115
  *
@@ -20,6 +20,25 @@ import { isReadOnlyDbError, loadFTSExtension } from './lbug-adapter.js';
20
20
  import { closeQueryResults } from './query-result-utils.js';
21
21
  import { createLbugDatabase, isWalCorruptionError, toNativeSafePath, WAL_RECOVERY_SUGGESTION, } from './lbug-config.js';
22
22
  import { guardWalQuarantine, isMissingFsError, isMissingShadowSidecarError, isReadOnlyShadowReplayError, preflightLbugSidecars, quarantineWalForMissingShadow, renameFailureMessage, statIfExists, } from './sidecar-recovery.js';
23
+ export async function statDbIdentity(dbPath) {
24
+ try {
25
+ const s = await fs.stat(dbPath);
26
+ return { ino: s.ino, mtimeMs: s.mtimeMs, size: s.size };
27
+ }
28
+ catch {
29
+ return null;
30
+ }
31
+ }
32
+ /** True only when both identities are known AND differ. A stat failure
33
+ * (ENOENT during the brief unlink window of a full rebuild) yields false, so
34
+ * the reader keeps serving its still-valid open inode until the NEW file
35
+ * appears with a different identity — avoiding a churn into a failed reopen
36
+ * mid-rebuild. */
37
+ export function dbIdentityChanged(prev, next) {
38
+ if (!prev || !next)
39
+ return false;
40
+ return prev.ino !== next.ino || prev.mtimeMs !== next.mtimeMs || prev.size !== next.size;
41
+ }
23
42
  const pool = new Map();
24
43
  const poolCloseListeners = new Set();
25
44
  /**
@@ -306,7 +325,16 @@ setInterval(() => {
306
325
  function createConnection(db) {
307
326
  silenceStdout();
308
327
  try {
309
- return new lbug.Connection(db);
328
+ const conn = new lbug.Connection(db);
329
+ // Bound a single query at the engine level so a pathological query cannot
330
+ // hang a pooled connection past the JS-side Promise.race guard (which frees
331
+ // the waiter but not the native call). Matches QUERY_TIMEOUT_MS. Guarded so
332
+ // test doubles that don't model the engine method don't break connection
333
+ // creation.
334
+ if (typeof conn.setQueryTimeout === 'function') {
335
+ conn.setQueryTimeout(QUERY_TIMEOUT_MS);
336
+ }
337
+ return conn;
310
338
  }
311
339
  finally {
312
340
  restoreStdout();
@@ -316,6 +344,8 @@ function createConnection(db) {
316
344
  const QUERY_TIMEOUT_MS = 30_000;
317
345
  /** Waiter queue timeout in milliseconds */
318
346
  const WAITER_TIMEOUT_MS = 15_000;
347
+ // Read-only open retry while `gitnexus analyze` writes. Catalogued as entry 4
348
+ // of the lbug-config retry-budget registry.
319
349
  const LOCK_RETRY_ATTEMPTS = 3;
320
350
  const LOCK_RETRY_DELAY_MS = 2000;
321
351
  const SHADOW_REPLAY_PROBE_QUERY = 'MATCH (n) RETURN n LIMIT 1';
@@ -502,18 +532,46 @@ const initPromises = new Map();
502
532
  * Concurrent calls for the same repoId are deduplicated — the second caller
503
533
  * awaits the first's in-progress init rather than starting a redundant one.
504
534
  */
535
+ /**
536
+ * Returns `true` when this call (re)opened a fresh handle onto the current
537
+ * on-disk file, `false` when it reused/served the existing handle (unchanged,
538
+ * or changed-but-a-query-is-in-flight). Callers that gate their own freshness
539
+ * bookkeeping on "did the pool actually roll over" (LocalBackend) use the
540
+ * return value; callers that only need the pool ready can ignore it.
541
+ */
505
542
  export const initLbug = async (repoId, dbPath) => {
506
543
  const existing = pool.get(repoId);
507
544
  if (existing) {
508
545
  existing.lastUsed = Date.now();
509
- return;
546
+ // Detect an index that `analyze` rebuilt or mutated under this live read
547
+ // pool. Without this, the pool keeps serving the old (POSIX:
548
+ // unlinked-but-open) inode until LRU/idle eviction — a stale-read window
549
+ // of up to IDLE_TIMEOUT_MS after analyze finishes.
550
+ const current = await statDbIdentity(dbPath);
551
+ if (!dbIdentityChanged(existing.dbIdentity, current))
552
+ return false; // unchanged → reuse
553
+ // A query is in flight on this entry; closing its connection (and the
554
+ // shared Database at refCount 0) mid-use is a native use-after-free. Serve
555
+ // the current handle for this dispatch — the next initLbug that finds the
556
+ // entry idle (checkedOut === 0) reopens, since the identity stays divergent
557
+ // until then. Under sustained overlapping queries `checkedOut` may never
558
+ // reach 0 and `lastUsed` keeps the idle timer from evicting, so this window
559
+ // is bounded by the load, not IDLE_TIMEOUT_MS — the data stays consistent
560
+ // (a complete older snapshot), just not the newest. Callers that route
561
+ // freshness THROUGH initLbug (rather than calling closeLbug directly) get
562
+ // this guard for free; that is why LocalBackend delegates here (#2614).
563
+ if (existing.checkedOut > 0)
564
+ return false;
565
+ closeOne(repoId); // idle & changed → evict, then fall through to reopen the new file
510
566
  }
511
567
  // Deduplicate concurrent init calls for the same repoId —
512
568
  // prevents double-init race when multiple parallel tool calls
513
569
  // trigger initialization for the same repo simultaneously.
514
570
  const pending = initPromises.get(repoId);
515
- if (pending)
516
- return pending;
571
+ if (pending) {
572
+ await pending;
573
+ return true;
574
+ }
517
575
  const promise = doInitLbug(repoId, dbPath);
518
576
  initPromises.set(repoId, promise);
519
577
  try {
@@ -522,6 +580,7 @@ export const initLbug = async (repoId, dbPath) => {
522
580
  finally {
523
581
  initPromises.delete(repoId);
524
582
  }
583
+ return true;
525
584
  };
526
585
  /**
527
586
  * Internal init — creates DB, pre-warms connections, loads FTS, then registers pool.
@@ -540,6 +599,21 @@ async function doInitLbug(repoId, dbPath) {
540
599
  // Reuse an existing native Database if another repoId already opened this path.
541
600
  // This prevents buffer manager exhaustion from multiple mmap regions on the same file.
542
601
  let shared = dbCache.get(dbPath);
602
+ if (shared && !shared.external && shared.dbIdentity) {
603
+ // #2614 F2: a cached read-only Database is keyed by dbPath and shared across
604
+ // pool consumers. If the on-disk index was rebuilt/swapped (new inode) while
605
+ // ANOTHER consumer still holds this handle (refCount kept it alive), reusing
606
+ // it serves a superseded index. Unreachable via the MCP backend (one
607
+ // consumer per lbugPath ⇒ refCount hits 0 ⇒ closeOne reopens fresh); a
608
+ // complete fix needs per-inode handles rather than a dbPath-keyed cache.
609
+ // Surface it so the corner is observable instead of silently stale.
610
+ const current = await statDbIdentity(dbPath);
611
+ if (dbIdentityChanged(shared.dbIdentity, current)) {
612
+ realStderrWrite(`GitNexus: reusing a shared read-only handle for ${dbPath} whose on-disk ` +
613
+ `index was rebuilt while another consumer holds it — results may be stale ` +
614
+ `until that consumer releases it.\n`);
615
+ }
616
+ }
543
617
  if (!shared) {
544
618
  // Open in read-only mode — MCP server never writes to the database.
545
619
  // This allows multiple MCP server instances to read concurrently, and
@@ -548,7 +622,7 @@ async function doInitLbug(repoId, dbPath) {
548
622
  for (let attempt = 1; attempt <= LOCK_RETRY_ATTEMPTS; attempt++) {
549
623
  try {
550
624
  const db = await openReadOnlyDatabase(dbPath);
551
- shared = { db, refCount: 0, ftsLoaded: false };
625
+ shared = { db, refCount: 0, ftsLoaded: false, dbIdentity: await statDbIdentity(dbPath) };
552
626
  dbCache.set(dbPath, shared);
553
627
  break;
554
628
  }
@@ -557,7 +631,12 @@ async function doInitLbug(repoId, dbPath) {
557
631
  if (isWalCorruptionError(lastError)) {
558
632
  try {
559
633
  const db = await tryQuarantineAndReopen(dbPath, repoId);
560
- shared = { db, refCount: 0, ftsLoaded: false };
634
+ shared = {
635
+ db,
636
+ refCount: 0,
637
+ ftsLoaded: false,
638
+ dbIdentity: await statDbIdentity(dbPath),
639
+ };
561
640
  dbCache.set(dbPath, shared);
562
641
  break;
563
642
  }
@@ -611,6 +690,9 @@ async function doInitLbug(repoId, dbPath) {
611
690
  // Register pool entry only after all connections are pre-warmed and FTS is
612
691
  // loaded. Concurrent executeQuery calls see either "not initialized"
613
692
  // (and throw cleanly) or a fully ready pool — never a half-built one.
693
+ // Record the on-disk identity so a later initLbug can detect an analyze
694
+ // rebuild/mutation and re-open onto the new file (pool staleness invalidation).
695
+ const dbIdentity = await statDbIdentity(dbPath);
614
696
  pool.set(repoId, {
615
697
  db,
616
698
  available,
@@ -618,6 +700,7 @@ async function doInitLbug(repoId, dbPath) {
618
700
  waiters: [],
619
701
  lastUsed: Date.now(),
620
702
  dbPath,
703
+ dbIdentity,
621
704
  closed: false,
622
705
  });
623
706
  ensureIdleTimer();
@@ -673,6 +756,8 @@ export async function initLbugWithDb(repoId, existingDb, dbPath) {
673
756
  waiters: [],
674
757
  lastUsed: Date.now(),
675
758
  dbPath,
759
+ // Injected/external DB (tests) — not tracked for rebuild invalidation.
760
+ dbIdentity: null,
676
761
  closed: false,
677
762
  });
678
763
  ensureIdleTimer();
@@ -97,6 +97,9 @@ export const runCheckpointWithRetry = async (options = {}) => {
97
97
  }
98
98
  }
99
99
  logger.warn({ attempts: CHECKPOINT_RETRY_ATTEMPTS }, 'GitNexus: manual WAL checkpoint exhausted retry budget — surfacing IO error to caller');
100
+ // The held-open cause (#2599) is named at the CLI layer (analyze.ts) where the
101
+ // --wal-checkpoint-threshold recovery hint already renders, so the original IO
102
+ // error is preserved intact for that classifier rather than re-wrapped here.
100
103
  throw lastError;
101
104
  };
102
105
  /**
@@ -10,6 +10,7 @@
10
10
  */
11
11
  import path from 'path';
12
12
  import fs from 'fs/promises';
13
+ import { retryRename } from '../storage/fs-atomic.js';
13
14
  import { runPipelineFromRepo } from './ingestion/pipeline.js';
14
15
  import { resetDegradedParseCounter } from './tree-sitter/safe-parse.js';
15
16
  import { initLbug, loadGraphToLbug, getLbugStats, executeQuery, executeWithReusedStatement, closeLbug, closeLbugBeforeExit, loadCachedEmbeddings, deleteNodesForFiles, deleteAllCommunitiesAndProcesses, deleteAllInterprocTaintPaths, deleteAllCallSummaries, deleteAllInjects, queryImportersBatch, loadFTSExtension, wipeLbugDbFiles, LbugWipeError, DELETE_FILES_CHUNK_SIZE, } from './lbug/lbug-adapter.js';
@@ -20,7 +21,7 @@ import { cjkSegmentationModeMismatch, getSearchFTSCjkSegmentation, initialiseSea
20
21
  import { getExtensionCapabilities, resolveAnalyzeInstallPolicy } from './lbug/extension-loader.js';
21
22
  import { diagnoseExtensionLoad } from './lbug/extension-load-error.js';
22
23
  import { startWalCheckpointDriver, checkpointOnce, } from './lbug/wal-checkpoint-driver.js';
23
- import { quarantineSidecarsForDirtyRecovery } from './lbug/sidecar-recovery.js';
24
+ import { quarantineSidecarsForDirtyRecovery, inspectLbugSidecars, } from './lbug/sidecar-recovery.js';
24
25
  import { getStoragePaths, resolveBranchPlacement, saveMeta, loadMeta, ensureGitNexusIgnored, registerRepo, adoptFlatBranchLabel, isReadOnlyFilesystemError, isRepoRegistered, cleanupOldKuzuFiles, reconcileMetadataFiles, isMissingFilesystemError, INDEX_METADATA_FILE, INCREMENTAL_SCHEMA_VERSION, } from '../storage/repo-manager.js';
25
26
  import { DEFAULT_PDG_MAX_FUNCTION_LINES } from './ingestion/cfg/collect.js';
26
27
  import { DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, } from './ingestion/cfg/emit.js';
@@ -901,6 +902,50 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
901
902
  const hashDiff = isIncremental
902
903
  ? diffFileHashes(newFileHashes, existingMeta.fileHashes)
903
904
  : undefined;
905
+ // #2 atomic index publish: on a full rebuild, build the fresh DB at a temp
906
+ // path and swap it over the live index in one rename at the very end, so a
907
+ // concurrent MCP reader opening mid-build only ever sees the previous
908
+ // complete index (never a wiped/half-built file) and a crash leaves the old
909
+ // index intact. The whole build flows through the singleton connection, so
910
+ // only initLbug/wipeLbugDbFiles below take the temp target.
911
+ //
912
+ // POSIX only: the common CLI/serve-worker analyze paths skip the native close
913
+ // (closeLbugBeforeExit, #2264) and leave the build handle open at swap time.
914
+ // POSIX renames an open file cleanly; a same-process open handle blocks the
915
+ // rename on Windows. Windows keeps the current in-place behavior
916
+ // (buildPath === lbugPath, no swap) until that is resolved (see §12/follow-up).
917
+ const isFullRebuild = !(isIncremental && hashDiff);
918
+ // Where the swap is allowed:
919
+ // - POSIX renames an open file, so the usual skip-native-close (#2264) is
920
+ // fine and the swap always applies.
921
+ // - Windows can swap only when a real close is safe to release the build
922
+ // handle before the rename — i.e. NOT a --pdg run (the #2264 destructor
923
+ // crash). Unverified on Windows CI; falls back to in-place otherwise.
924
+ const posixSwap = process.platform !== 'win32';
925
+ // #2614 Windows: the forced real-close before the rename re-bets that #2264 is
926
+ // --pdg-only, which is unproven (the CLI/worker skip the native close
927
+ // UNCONDITIONALLY) and unverifiable without a Windows runner. Keep it opt-in
928
+ // (GITNEXUS_ATOMIC_WINDOWS_SWAP=1) so the default Windows analyze stays on the
929
+ // proven in-place path; enable it only to test the Windows swap.
930
+ const windowsSwapOk = process.platform === 'win32' &&
931
+ options.pdg !== true &&
932
+ process.env.GITNEXUS_ATOMIC_WINDOWS_SWAP === '1';
933
+ // Incremental atomicity copies the whole index into the temp before mutating
934
+ // it, which negates incremental's speed premise — so it is opt-in
935
+ // (GITNEXUS_ATOMIC_INCREMENTAL=1) pending a benchmark. Full rebuilds always
936
+ // swap where the platform allows.
937
+ const wantAtomicIncremental = isIncremental && !!hashDiff && process.env.GITNEXUS_ATOMIC_INCREMENTAL === '1';
938
+ // #2614 F3: the copy-then-swap stages ONLY the main lbug file, so a live index
939
+ // carrying an orphan .wal/.shadow (a silently-failed prior checkpoint) would
940
+ // be copied incompletely and lose that delta. Only take the atomic path when
941
+ // the live index is a consolidated single file; otherwise fall back to the
942
+ // in-place writeback, which the next open replays correctly.
943
+ const atomicIncremental = wantAtomicIncremental && (await inspectLbugSidecars(lbugPath)).kind === 'clean';
944
+ if (wantAtomicIncremental && !atomicIncremental) {
945
+ log('atomic-incremental: live index carries orphan sidecars — using in-place writeback');
946
+ }
947
+ const useAtomicSwap = (isFullRebuild || atomicIncremental) && (posixSwap || windowsSwapOk);
948
+ const buildPath = useAtomicSwap ? `${lbugPath}.new` : lbugPath;
904
949
  if (isIncremental && hashDiff) {
905
950
  log(`Incremental: changed=${hashDiff.changed.length}, ` +
906
951
  `added=${hashDiff.added.length}, ` +
@@ -919,6 +964,14 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
919
964
  directWriteCount: hashDiff.toWrite.length,
920
965
  },
921
966
  });
967
+ if (atomicIncremental) {
968
+ // Stage the live index into the temp so the in-place delete/writeback
969
+ // below mutates the COPY, and the end-of-run swap publishes it atomically.
970
+ // Clear any stale temp first (a crashed run), then copy the (consolidated,
971
+ // single-file) live index. Whole-file copy — hence opt-in.
972
+ await wipeLbugDbFiles(buildPath);
973
+ await fs.copyFile(lbugPath, buildPath);
974
+ }
922
975
  }
923
976
  else {
924
977
  // Full rebuild path: wipe DB files first.
@@ -949,7 +1002,12 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
949
1002
  // valve below can never drift. Failures now throw a typed LbugWipeError
950
1003
  // (ENOENT-verified removal) instead of silently letting initLbug reopen
951
1004
  // a still-populated DB this run believes it wiped.
952
- await wipeLbugDbFiles(lbugPath);
1005
+ //
1006
+ // With the atomic swap (POSIX), this wipes the TEMP build target
1007
+ // (`buildPath` = `<lbugPath>.new`, clearing any stragglers from a crashed
1008
+ // run) and leaves the live index untouched until the end-of-run swap. On
1009
+ // Windows buildPath === lbugPath, so this is the original in-place wipe.
1010
+ await wipeLbugDbFiles(buildPath);
953
1011
  }
954
1012
  // Size the buffer pool to the graph just built by the pipeline (a page cache
955
1013
  // over the on-disk index, which scales with node/edge count) instead of the
@@ -958,7 +1016,9 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
958
1016
  // the pool; env override / no-hint paths are unchanged. See
959
1017
  // resolveBufferManagerSize / estimateBufferPool.
960
1018
  setBufferPoolSizeHint(estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount));
961
- await initLbug(lbugPath);
1019
+ // Full rebuild (POSIX) builds into the temp `buildPath`; incremental and
1020
+ // Windows use `buildPath === lbugPath` in place.
1021
+ await initLbug(buildPath);
962
1022
  // Manual WAL checkpoint driver (#1741): periodically drain the WAL
963
1023
  // from JS so the un-retriable native auto-checkpoint almost never
964
1024
  // has work left to do. Failures of the manual CHECKPOINT are absorbed
@@ -1191,8 +1251,8 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
1191
1251
  // to replace wholesale.
1192
1252
  await walCheckpointDriver.stop();
1193
1253
  await closeLbug();
1194
- await wipeLbugDbFiles(lbugPath);
1195
- await initLbug(lbugPath);
1254
+ await wipeLbugDbFiles(buildPath);
1255
+ await initLbug(buildPath);
1196
1256
  walCheckpointDriver = startWalCheckpointDriver();
1197
1257
  await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
1198
1258
  lbugMsgCount++;
@@ -1726,7 +1786,11 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
1726
1786
  // inside the resolver, and a mismatch leaves the dirty flag intact so the
1727
1787
  // next run takes the established full-recovery path.
1728
1788
  meta.runnerIdentity = finalizeAnalyzerRunnerIdentity(import.meta.url, runnerIdentity);
1729
- await saveMeta(metaDir, meta);
1789
+ // #2614 F1: the freshness stamp (saveMeta) is written AFTER the atomic swap
1790
+ // below — never here — so a concurrent MCP reader can't observe
1791
+ // meta.indexedAt = T_new while lbugPath still resolves to the pre-swap
1792
+ // inode (which latched the reader on the stale index permanently). The meta
1793
+ // object is fully computed at this point; only its write is deferred.
1730
1794
  // Persist the incremental parse cache for the next run. Wraps in
1731
1795
  // try/catch so a cache-write failure never breaks an otherwise
1732
1796
  // successful indexing run. Prune stale chunk-hash entries first so
@@ -1850,7 +1914,54 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
1850
1914
  // LadybugDB destructor double-free after --pdg writes — closeLbugBeforeExit
1851
1915
  // CHECKPOINTs for durability then leaves the handles for process exit to
1852
1916
  // reclaim (#2264). Long-lived callers close for real.
1853
- await (options.skipNativeCloseOnExit ? closeLbugBeforeExit() : closeLbug());
1917
+ //
1918
+ // On Windows a swap must release the build handle before the rename (a
1919
+ // same-process open file can't be renamed), so it forces a real close —
1920
+ // safe because windowsSwapOk excludes --pdg (the #2264 case). POSIX renames
1921
+ // an open file, so it keeps the skip-native-close there.
1922
+ const forceRealCloseForSwap = useAtomicSwap && process.platform === 'win32';
1923
+ await (options.skipNativeCloseOnExit && !forceRealCloseForSwap
1924
+ ? closeLbugBeforeExit()
1925
+ : closeLbug());
1926
+ // #2 atomic publish: the fresh index was built at buildPath (a full rebuild,
1927
+ // or an opt-in atomic incremental that copied the live index in first). Swap
1928
+ // it over the live lbugPath in one rename so an MCP reader that opened
1929
+ // mid-build only ever saw the previous complete index — never a wiped/
1930
+ // half-built file. The close above checkpoint-consolidated buildPath to a
1931
+ // single file (no .wal), so the rename publishes a complete index; a reader
1932
+ // holding the old inode keeps a consistent stale snapshot until the pool
1933
+ // re-opens onto the new one (the pool staleness invalidation). Runs only on
1934
+ // success — a thrown error skips this, leaving the live index intact and the
1935
+ // temp build to be cleared by the next run's wipe.
1936
+ // Only publish if the build actually produced a DB at buildPath. A
1937
+ // degenerate run (empty repo, or a mocked pipeline that never opened the
1938
+ // store) leaves nothing to swap — skip rather than throw ENOENT.
1939
+ const builtDbExists = useAtomicSwap
1940
+ ? await fs.stat(buildPath).then(() => true, () => false)
1941
+ : false;
1942
+ if (useAtomicSwap && builtDbExists) {
1943
+ await retryRename(buildPath, lbugPath);
1944
+ // Clear any sidecars orphaned beside the replaced file. A cleanly-closed
1945
+ // prior index has none; a crashed one could, and it would be replay
1946
+ // poison next to the freshly published index. Best-effort.
1947
+ for (const suffix of ['.wal', '.shadow', '.wal.checkpoint']) {
1948
+ await fs.rm(`${lbugPath}${suffix}`, { force: true }).catch(() => { });
1949
+ }
1950
+ // #2614 F4: if the final checkpoint silently failed, the build may still
1951
+ // carry a residual .wal/.shadow under the temp name. MOVE it beside the
1952
+ // published index (not orphan/delete it) so the next open replays the
1953
+ // delta, rather than leaving it under a name LadybugDB never reconciles.
1954
+ for (const suffix of ['.wal', '.shadow']) {
1955
+ await fs.rename(`${buildPath}${suffix}`, `${lbugPath}${suffix}`).catch(() => { });
1956
+ }
1957
+ }
1958
+ // #2614 F1: stamp the freshness metadata now that the index is published.
1959
+ // When meta.indexedAt becomes visible, lbugPath already resolves to the new
1960
+ // inode, so a reader reiniting on the stamp opens the fresh graph rather
1961
+ // than latching on the old one. Leaving the dirty flag set across the swap
1962
+ // is a crash-safety improvement: a failed swap leaves the previous index
1963
+ // live and the next run recovers via the full-rebuild path.
1964
+ await saveMeta(metaDir, meta);
1854
1965
  progress('done', 100, 'Done');
1855
1966
  return {
1856
1967
  repoName: projectName,
@@ -147,6 +147,7 @@ export declare class LocalBackend {
147
147
  private reinitPromises;
148
148
  private lastStalenessCheck;
149
149
  private lastObservedIndexedAt;
150
+ private lastObservedDbIdentity;
150
151
  private groupToolSvc;
151
152
  /**
152
153
  * One-shot stderr warnings for sibling-clone drift, keyed by
@@ -8,7 +8,7 @@
8
8
  import fs from 'fs/promises';
9
9
  import path from 'path';
10
10
  import { createHash } from 'crypto';
11
- import { initLbug, executeQuery, executeParameterized, closeLbug, isLbugReady, } from '../../core/lbug/pool-adapter.js';
11
+ import { initLbug, executeQuery, executeParameterized, closeLbug, isLbugReady, statDbIdentity, dbIdentityChanged, } from '../../core/lbug/pool-adapter.js';
12
12
  import { queryClassBeanMetadata } from './bean-metadata.js';
13
13
  import { isValidQueryParams } from '../../core/lbug/query-params.js';
14
14
  import { toDisplayLine } from './line-display.js';
@@ -470,6 +470,11 @@ export class LocalBackend {
470
470
  // not persist across calls and the staleness check would reinit forever
471
471
  // (#2106).
472
472
  lastObservedIndexedAt = new Map();
473
+ // #2614 F1: file identity of the lbug the pool last opened. An atomic swap or
474
+ // an in-place incremental changes the inode; reiniting on that reinit-covers
475
+ // the window where meta.indexedAt hasn't caught up (and the incremental case),
476
+ // so a rebuilt index is never served stale even when the stamp looks current.
477
+ lastObservedDbIdentity = new Map();
473
478
  groupToolSvc = null;
474
479
  /**
475
480
  * One-shot stderr warnings for sibling-clone drift, keyed by
@@ -755,6 +760,7 @@ export class LocalBackend {
755
760
  this.initializedRepos.delete(key);
756
761
  this.lastStalenessCheck.delete(key);
757
762
  this.lastObservedIndexedAt.delete(key);
763
+ this.lastObservedDbIdentity.delete(key);
758
764
  this.reinitPromises.delete(key);
759
765
  closeLbug(key).catch(() => { });
760
766
  }
@@ -1114,23 +1120,38 @@ export class LocalBackend {
1114
1120
  // Reading the flat meta for a branch handle would compare the branch
1115
1121
  // index's indexedAt against the primary's and thrash the pool (#2106).
1116
1122
  const meta = await loadMeta(path.dirname(repo.lbugPath));
1117
- if (!meta)
1118
- return;
1119
1123
  // Compare against the last indexedAt OBSERVED for this pool (keyed by
1120
1124
  // lbugPath), not the handle's — branch handles are fresh spreads so a
1121
1125
  // handle mutation would not persist and would reinit on every check.
1122
1126
  const observed = this.lastObservedIndexedAt.get(poolKey) ?? repo.indexedAt;
1123
- if (meta.indexedAt && meta.indexedAt !== observed) {
1124
- // Index was rebuilt — close stale connection and re-init.
1125
- // Wrap in reinitPromises to prevent TOCTOU race where concurrent
1126
- // callers both detect staleness and double-close the pool.
1127
+ const stampChanged = !!meta?.indexedAt && meta.indexedAt !== observed;
1128
+ // #2614 F1: also reinit on a file-identity change. An atomic swap (or an
1129
+ // in-place incremental) changes the lbug inode; keying only on
1130
+ // meta.indexedAt let a reader that reinited inside the pre-swap window
1131
+ // latch on the old inode forever (its stamp already == meta.indexedAt).
1132
+ const currentIdentity = await statDbIdentity(repo.lbugPath);
1133
+ const identityChanged = dbIdentityChanged(this.lastObservedDbIdentity.get(poolKey) ?? null, currentIdentity);
1134
+ if (stampChanged || identityChanged) {
1135
+ // Index was rebuilt/swapped — DELEGATE the close/reopen to the pool's
1136
+ // initLbug, which refuses to evict (and close the shared Database)
1137
+ // while a query is in flight (its checkedOut>0 guard). Calling
1138
+ // closeLbug directly here bypassed that guard and could close a
1139
+ // Database mid-query — a native use-after-free (#2614). Wrap in
1140
+ // reinitPromises to serialize concurrent detectors.
1127
1141
  const reinit = (async () => {
1128
1142
  try {
1129
- await closeLbug(poolKey);
1130
- this.initializedRepos.delete(poolKey);
1131
- this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
1132
- await initLbug(poolKey, repo.lbugPath);
1133
- this.initializedRepos.add(poolKey);
1143
+ // Advance the observed stamp regardless: a stamp change with an
1144
+ // unchanged file must not re-trigger on every check.
1145
+ if (meta?.indexedAt)
1146
+ this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
1147
+ const reopened = await initLbug(poolKey, repo.lbugPath);
1148
+ // Advance the observed IDENTITY only when the pool actually rolled
1149
+ // over. If a query was in flight, initLbug served the current
1150
+ // handle and returned false; leaving the identity divergent
1151
+ // re-triggers the reopen on a later idle check instead of latching.
1152
+ if (reopened) {
1153
+ this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
1154
+ }
1134
1155
  }
1135
1156
  finally {
1136
1157
  this.reinitPromises.delete(poolKey);
@@ -1151,6 +1172,7 @@ export class LocalBackend {
1151
1172
  await initLbug(poolKey, repo.lbugPath);
1152
1173
  this.initializedRepos.add(poolKey);
1153
1174
  this.lastObservedIndexedAt.set(poolKey, repo.indexedAt);
1175
+ this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
1154
1176
  }
1155
1177
  catch (err) {
1156
1178
  // If lock error, mark as not initialized so next call retries
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "gitnexus",
3
- "version": "1.6.10-rc.80",
3
+ "version": "1.6.10-rc.82",
4
4
  "description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
5
5
  "author": "Abhigyan Patwari",
6
6
  "license": "PolyForm-Noncommercial-1.0.0",