gitnexus 1.6.10-rc.80 → 1.6.10-rc.82
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/analyze.js +9 -1
- package/dist/core/group/bridge-db.js +2 -0
- package/dist/core/lbug/lbug-config.d.ts +9 -0
- package/dist/core/lbug/lbug-config.js +33 -0
- package/dist/core/lbug/pool-adapter.d.ts +23 -1
- package/dist/core/lbug/pool-adapter.js +91 -6
- package/dist/core/lbug/wal-checkpoint-driver.js +3 -0
- package/dist/core/run-analyze.js +118 -7
- package/dist/mcp/local/local-backend.d.ts +1 -0
- package/dist/mcp/local/local-backend.js +34 -12
- package/package.json +1 -1
package/dist/cli/analyze.js
CHANGED
|
@@ -14,7 +14,7 @@ import v8 from 'v8';
|
|
|
14
14
|
import cliProgress from 'cli-progress';
|
|
15
15
|
import { isLbugReady, LbugWipeError } from '../core/lbug/lbug-adapter.js';
|
|
16
16
|
import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js';
|
|
17
|
-
import { getOsPageSize, isLbugCheckpointIoError, isLbugPageSizeFrameError, isPageSizeAwareLadybug, isWalCorruptionError, parseWalCheckpointThreshold, WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js';
|
|
17
|
+
import { getOsPageSize, isLbugCheckpointIoError, isLbugCheckpointBusyError, isLbugPageSizeFrameError, isPageSizeAwareLadybug, isWalCorruptionError, parseWalCheckpointThreshold, WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js';
|
|
18
18
|
import { getStoragePaths, getGlobalRegistryPath, RegistryNameCollisionError, AnalysisNotFinalizedError, assertAnalysisFinalized, } from '../storage/repo-manager.js';
|
|
19
19
|
import { getGitRoot, hasGitDir, getDefaultBranch } from '../storage/git.js';
|
|
20
20
|
import { loadAnalyzeConfig, mergeAnalyzeOptions, resolveDefaultBranch, validateBranchName, GitNexusRcError, } from './analyze-config.js';
|
|
@@ -1236,7 +1236,15 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
|
|
|
1236
1236
|
return;
|
|
1237
1237
|
}
|
|
1238
1238
|
if (isLbugCheckpointIoError(err)) {
|
|
1239
|
+
// #2599: when the checkpoint IO error also looks busy/locked, another
|
|
1240
|
+
// handle holds the store open — name that actionable cause alongside the
|
|
1241
|
+
// threshold hint (the original error is preserved so the hint still fires).
|
|
1242
|
+
const heldOpen = isLbugCheckpointBusyError(err)
|
|
1243
|
+
? ` Another process may hold the store open (a running \`gitnexus mcp\` server, or a\n` +
|
|
1244
|
+
` stale reader) — close other GitNexus processes on this repo, then retry.\n`
|
|
1245
|
+
: '';
|
|
1239
1246
|
cliError(` LadybugDB failed while rotating/removing WAL checkpoint files.\n` +
|
|
1247
|
+
heldOpen +
|
|
1240
1248
|
` This can happen when auto-checkpoint runs at the default threshold (~16MB).\n` +
|
|
1241
1249
|
` Retry with a larger checkpoint threshold to reduce checkpoint frequency:\n` +
|
|
1242
1250
|
` gitnexus analyze --wal-checkpoint-threshold ${RECOMMENDED_WAL_CHECKPOINT_THRESHOLD}\n` +
|
|
@@ -829,6 +829,8 @@ const LBUG_OPEN_RETRY_PATTERNS = [
|
|
|
829
829
|
'could not set lock',
|
|
830
830
|
'lock held by another process',
|
|
831
831
|
];
|
|
832
|
+
// Cross-repo bridge RO open retry. Catalogued as entry 5 of the lbug-config
|
|
833
|
+
// retry-budget registry; caps back-off so total wait ~3s.
|
|
832
834
|
const LBUG_OPEN_RETRY_ATTEMPTS = 10;
|
|
833
835
|
const LBUG_OPEN_RETRY_BASE_MS = 100;
|
|
834
836
|
/** Cap individual back-off delays so the total wait is bounded (~3s). */
|
|
@@ -128,6 +128,15 @@ export interface LbugConnectionHandle {
|
|
|
128
128
|
* import directly from this module — no re-export to keep in sync.
|
|
129
129
|
*/
|
|
130
130
|
export declare const isDbBusyError: (err: unknown) => boolean;
|
|
131
|
+
/**
|
|
132
|
+
* True when a WAL-checkpoint IO error ALSO carries a busy/lock signal — the
|
|
133
|
+
* rotation failed because another handle (a `gitnexus mcp` server, or this
|
|
134
|
+
* process's own reader) holds the store's WAL open, rather than a permanent
|
|
135
|
+
* disk error. Reuses `isDbBusyError`'s already-tested keyword set instead of a
|
|
136
|
+
* fresh regex, so an unmatched message degrades to "IO error" rather than
|
|
137
|
+
* silently claiming a held-open cause. (#2599)
|
|
138
|
+
*/
|
|
139
|
+
export declare const isLbugCheckpointBusyError: (err: unknown) => boolean;
|
|
131
140
|
/** See {@link classifyDeleteAllError}. */
|
|
132
141
|
export type DeleteAllErrorClass = 'benign-missing-table' | 'rethrow';
|
|
133
142
|
/**
|
|
@@ -582,6 +582,27 @@ export const isDbBusyError = (err) => {
|
|
|
582
582
|
msg.includes('already in use') ||
|
|
583
583
|
msg.includes('only one write transaction at a time'));
|
|
584
584
|
};
|
|
585
|
+
/**
|
|
586
|
+
* True when a WAL-checkpoint IO error ALSO carries a busy/lock signal — the
|
|
587
|
+
* rotation failed because another handle (a `gitnexus mcp` server, or this
|
|
588
|
+
* process's own reader) holds the store's WAL open, rather than a permanent
|
|
589
|
+
* disk error. Reuses `isDbBusyError`'s already-tested keyword set instead of a
|
|
590
|
+
* fresh regex, so an unmatched message degrades to "IO error" rather than
|
|
591
|
+
* silently claiming a held-open cause. (#2599)
|
|
592
|
+
*/
|
|
593
|
+
export const isLbugCheckpointBusyError = (err) => {
|
|
594
|
+
if (!isLbugCheckpointIoError(err))
|
|
595
|
+
return false;
|
|
596
|
+
// Anchor to real held-open wording rather than isDbBusyError's broad
|
|
597
|
+
// `.includes('lock')`, which matches the DB PATH embedded in the checkpoint
|
|
598
|
+
// error message (e.g. a repo under `blockchain-app`) and would misclassify a
|
|
599
|
+
// pure disk fault as held-open (#2614 LOW).
|
|
600
|
+
const msg = (err instanceof Error ? err.message : String(err)).toLowerCase();
|
|
601
|
+
return (msg.includes('could not set lock') ||
|
|
602
|
+
msg.includes('lock is held') ||
|
|
603
|
+
msg.includes('being used by another process') ||
|
|
604
|
+
msg.includes('is busy'));
|
|
605
|
+
};
|
|
585
606
|
/**
|
|
586
607
|
* Classify an error thrown while clearing all relationships of one type
|
|
587
608
|
* before an incremental re-write (`deleteAllRelationshipsOfType` in
|
|
@@ -652,6 +673,18 @@ const OPEN_LOCK_RETRY_DELAY_MS = 100;
|
|
|
652
673
|
export const HANDLE_RELEASE_PROBE_ATTEMPTS = 5;
|
|
653
674
|
export const HANDLE_RELEASE_PROBE_DELAY_MS = 50;
|
|
654
675
|
const HANDLE_RELEASE_LOCK_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']);
|
|
676
|
+
// Retry-budget registry, part 2 (retry-budget consolidation): the remaining
|
|
677
|
+
// open-time lock retries live next to their call sites but are catalogued here
|
|
678
|
+
// so all lbug retry budgets surface in one grep. They retry the same lock class
|
|
679
|
+
// as 1–3 ("Could not set lock" while a writer rebuilds the index):
|
|
680
|
+
// 4. LOCK_RETRY_ATTEMPTS / LOCK_RETRY_DELAY_MS (pool-adapter.ts)
|
|
681
|
+
// → read pool's read-only open while `gitnexus analyze` is writing
|
|
682
|
+
// (3 attempts, linear 2s·n back-off ≈ 6s total)
|
|
683
|
+
// 5. LBUG_OPEN_RETRY_ATTEMPTS / _BASE_MS / _MAX_MS (group/bridge-db.ts)
|
|
684
|
+
// → cross-repo bridge RO open race (10 attempts, linear 100ms·n capped
|
|
685
|
+
// at 500ms ≈ 3.5s total)
|
|
686
|
+
// Kept in-file (not moved here) so explicit `lbug-config` test mocks don't have
|
|
687
|
+
// to enumerate them; change a budget in its call site and update this catalogue.
|
|
655
688
|
/**
|
|
656
689
|
* Test-fixture directory prefixes recognized by `isTestFixturePath`.
|
|
657
690
|
*
|
|
@@ -15,6 +15,21 @@
|
|
|
15
15
|
* from the same Database is the officially supported concurrency pattern.
|
|
16
16
|
*/
|
|
17
17
|
import lbug from '@ladybugdb/core';
|
|
18
|
+
/** Filesystem identity used to detect an index rebuilt/mutated under a live
|
|
19
|
+
* read pool. `ino` catches a full-rebuild unlink+recreate or an atomic-rename
|
|
20
|
+
* swap; `mtimeMs`+`size` catch an in-place incremental writeback. */
|
|
21
|
+
interface DbIdentity {
|
|
22
|
+
ino: number;
|
|
23
|
+
mtimeMs: number;
|
|
24
|
+
size: number;
|
|
25
|
+
}
|
|
26
|
+
export declare function statDbIdentity(dbPath: string): Promise<DbIdentity | null>;
|
|
27
|
+
/** True only when both identities are known AND differ. A stat failure
|
|
28
|
+
* (ENOENT during the brief unlink window of a full rebuild) yields false, so
|
|
29
|
+
* the reader keeps serving its still-valid open inode until the NEW file
|
|
30
|
+
* appears with a different identity — avoiding a churn into a failed reopen
|
|
31
|
+
* mid-rebuild. */
|
|
32
|
+
export declare function dbIdentityChanged(prev: DbIdentity | null, next: DbIdentity | null): boolean;
|
|
18
33
|
/**
|
|
19
34
|
* Listeners notified when a pool entry is torn down (LRU eviction, idle
|
|
20
35
|
* timeout, explicit close). Used by upper layers (e.g. the BM25 search
|
|
@@ -87,7 +102,14 @@ export declare function restoreStdout(): void;
|
|
|
87
102
|
* Concurrent calls for the same repoId are deduplicated — the second caller
|
|
88
103
|
* awaits the first's in-progress init rather than starting a redundant one.
|
|
89
104
|
*/
|
|
90
|
-
|
|
105
|
+
/**
|
|
106
|
+
* Returns `true` when this call (re)opened a fresh handle onto the current
|
|
107
|
+
* on-disk file, `false` when it reused/served the existing handle (unchanged,
|
|
108
|
+
* or changed-but-a-query-is-in-flight). Callers that gate their own freshness
|
|
109
|
+
* bookkeeping on "did the pool actually roll over" (LocalBackend) use the
|
|
110
|
+
* return value; callers that only need the pool ready can ignore it.
|
|
111
|
+
*/
|
|
112
|
+
export declare const initLbug: (repoId: string, dbPath: string) => Promise<boolean>;
|
|
91
113
|
/**
|
|
92
114
|
* Initialize a pool entry from a pre-existing Database object.
|
|
93
115
|
*
|
|
@@ -20,6 +20,25 @@ import { isReadOnlyDbError, loadFTSExtension } from './lbug-adapter.js';
|
|
|
20
20
|
import { closeQueryResults } from './query-result-utils.js';
|
|
21
21
|
import { createLbugDatabase, isWalCorruptionError, toNativeSafePath, WAL_RECOVERY_SUGGESTION, } from './lbug-config.js';
|
|
22
22
|
import { guardWalQuarantine, isMissingFsError, isMissingShadowSidecarError, isReadOnlyShadowReplayError, preflightLbugSidecars, quarantineWalForMissingShadow, renameFailureMessage, statIfExists, } from './sidecar-recovery.js';
|
|
23
|
+
export async function statDbIdentity(dbPath) {
|
|
24
|
+
try {
|
|
25
|
+
const s = await fs.stat(dbPath);
|
|
26
|
+
return { ino: s.ino, mtimeMs: s.mtimeMs, size: s.size };
|
|
27
|
+
}
|
|
28
|
+
catch {
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
/** True only when both identities are known AND differ. A stat failure
|
|
33
|
+
* (ENOENT during the brief unlink window of a full rebuild) yields false, so
|
|
34
|
+
* the reader keeps serving its still-valid open inode until the NEW file
|
|
35
|
+
* appears with a different identity — avoiding a churn into a failed reopen
|
|
36
|
+
* mid-rebuild. */
|
|
37
|
+
export function dbIdentityChanged(prev, next) {
|
|
38
|
+
if (!prev || !next)
|
|
39
|
+
return false;
|
|
40
|
+
return prev.ino !== next.ino || prev.mtimeMs !== next.mtimeMs || prev.size !== next.size;
|
|
41
|
+
}
|
|
23
42
|
const pool = new Map();
|
|
24
43
|
const poolCloseListeners = new Set();
|
|
25
44
|
/**
|
|
@@ -306,7 +325,16 @@ setInterval(() => {
|
|
|
306
325
|
function createConnection(db) {
|
|
307
326
|
silenceStdout();
|
|
308
327
|
try {
|
|
309
|
-
|
|
328
|
+
const conn = new lbug.Connection(db);
|
|
329
|
+
// Bound a single query at the engine level so a pathological query cannot
|
|
330
|
+
// hang a pooled connection past the JS-side Promise.race guard (which frees
|
|
331
|
+
// the waiter but not the native call). Matches QUERY_TIMEOUT_MS. Guarded so
|
|
332
|
+
// test doubles that don't model the engine method don't break connection
|
|
333
|
+
// creation.
|
|
334
|
+
if (typeof conn.setQueryTimeout === 'function') {
|
|
335
|
+
conn.setQueryTimeout(QUERY_TIMEOUT_MS);
|
|
336
|
+
}
|
|
337
|
+
return conn;
|
|
310
338
|
}
|
|
311
339
|
finally {
|
|
312
340
|
restoreStdout();
|
|
@@ -316,6 +344,8 @@ function createConnection(db) {
|
|
|
316
344
|
const QUERY_TIMEOUT_MS = 30_000;
|
|
317
345
|
/** Waiter queue timeout in milliseconds */
|
|
318
346
|
const WAITER_TIMEOUT_MS = 15_000;
|
|
347
|
+
// Read-only open retry while `gitnexus analyze` writes. Catalogued as entry 4
|
|
348
|
+
// of the lbug-config retry-budget registry.
|
|
319
349
|
const LOCK_RETRY_ATTEMPTS = 3;
|
|
320
350
|
const LOCK_RETRY_DELAY_MS = 2000;
|
|
321
351
|
const SHADOW_REPLAY_PROBE_QUERY = 'MATCH (n) RETURN n LIMIT 1';
|
|
@@ -502,18 +532,46 @@ const initPromises = new Map();
|
|
|
502
532
|
* Concurrent calls for the same repoId are deduplicated — the second caller
|
|
503
533
|
* awaits the first's in-progress init rather than starting a redundant one.
|
|
504
534
|
*/
|
|
535
|
+
/**
|
|
536
|
+
* Returns `true` when this call (re)opened a fresh handle onto the current
|
|
537
|
+
* on-disk file, `false` when it reused/served the existing handle (unchanged,
|
|
538
|
+
* or changed-but-a-query-is-in-flight). Callers that gate their own freshness
|
|
539
|
+
* bookkeeping on "did the pool actually roll over" (LocalBackend) use the
|
|
540
|
+
* return value; callers that only need the pool ready can ignore it.
|
|
541
|
+
*/
|
|
505
542
|
export const initLbug = async (repoId, dbPath) => {
|
|
506
543
|
const existing = pool.get(repoId);
|
|
507
544
|
if (existing) {
|
|
508
545
|
existing.lastUsed = Date.now();
|
|
509
|
-
|
|
546
|
+
// Detect an index that `analyze` rebuilt or mutated under this live read
|
|
547
|
+
// pool. Without this, the pool keeps serving the old (POSIX:
|
|
548
|
+
// unlinked-but-open) inode until LRU/idle eviction — a stale-read window
|
|
549
|
+
// of up to IDLE_TIMEOUT_MS after analyze finishes.
|
|
550
|
+
const current = await statDbIdentity(dbPath);
|
|
551
|
+
if (!dbIdentityChanged(existing.dbIdentity, current))
|
|
552
|
+
return false; // unchanged → reuse
|
|
553
|
+
// A query is in flight on this entry; closing its connection (and the
|
|
554
|
+
// shared Database at refCount 0) mid-use is a native use-after-free. Serve
|
|
555
|
+
// the current handle for this dispatch — the next initLbug that finds the
|
|
556
|
+
// entry idle (checkedOut === 0) reopens, since the identity stays divergent
|
|
557
|
+
// until then. Under sustained overlapping queries `checkedOut` may never
|
|
558
|
+
// reach 0 and `lastUsed` keeps the idle timer from evicting, so this window
|
|
559
|
+
// is bounded by the load, not IDLE_TIMEOUT_MS — the data stays consistent
|
|
560
|
+
// (a complete older snapshot), just not the newest. Callers that route
|
|
561
|
+
// freshness THROUGH initLbug (rather than calling closeLbug directly) get
|
|
562
|
+
// this guard for free; that is why LocalBackend delegates here (#2614).
|
|
563
|
+
if (existing.checkedOut > 0)
|
|
564
|
+
return false;
|
|
565
|
+
closeOne(repoId); // idle & changed → evict, then fall through to reopen the new file
|
|
510
566
|
}
|
|
511
567
|
// Deduplicate concurrent init calls for the same repoId —
|
|
512
568
|
// prevents double-init race when multiple parallel tool calls
|
|
513
569
|
// trigger initialization for the same repo simultaneously.
|
|
514
570
|
const pending = initPromises.get(repoId);
|
|
515
|
-
if (pending)
|
|
516
|
-
|
|
571
|
+
if (pending) {
|
|
572
|
+
await pending;
|
|
573
|
+
return true;
|
|
574
|
+
}
|
|
517
575
|
const promise = doInitLbug(repoId, dbPath);
|
|
518
576
|
initPromises.set(repoId, promise);
|
|
519
577
|
try {
|
|
@@ -522,6 +580,7 @@ export const initLbug = async (repoId, dbPath) => {
|
|
|
522
580
|
finally {
|
|
523
581
|
initPromises.delete(repoId);
|
|
524
582
|
}
|
|
583
|
+
return true;
|
|
525
584
|
};
|
|
526
585
|
/**
|
|
527
586
|
* Internal init — creates DB, pre-warms connections, loads FTS, then registers pool.
|
|
@@ -540,6 +599,21 @@ async function doInitLbug(repoId, dbPath) {
|
|
|
540
599
|
// Reuse an existing native Database if another repoId already opened this path.
|
|
541
600
|
// This prevents buffer manager exhaustion from multiple mmap regions on the same file.
|
|
542
601
|
let shared = dbCache.get(dbPath);
|
|
602
|
+
if (shared && !shared.external && shared.dbIdentity) {
|
|
603
|
+
// #2614 F2: a cached read-only Database is keyed by dbPath and shared across
|
|
604
|
+
// pool consumers. If the on-disk index was rebuilt/swapped (new inode) while
|
|
605
|
+
// ANOTHER consumer still holds this handle (refCount kept it alive), reusing
|
|
606
|
+
// it serves a superseded index. Unreachable via the MCP backend (one
|
|
607
|
+
// consumer per lbugPath ⇒ refCount hits 0 ⇒ closeOne reopens fresh); a
|
|
608
|
+
// complete fix needs per-inode handles rather than a dbPath-keyed cache.
|
|
609
|
+
// Surface it so the corner is observable instead of silently stale.
|
|
610
|
+
const current = await statDbIdentity(dbPath);
|
|
611
|
+
if (dbIdentityChanged(shared.dbIdentity, current)) {
|
|
612
|
+
realStderrWrite(`GitNexus: reusing a shared read-only handle for ${dbPath} whose on-disk ` +
|
|
613
|
+
`index was rebuilt while another consumer holds it — results may be stale ` +
|
|
614
|
+
`until that consumer releases it.\n`);
|
|
615
|
+
}
|
|
616
|
+
}
|
|
543
617
|
if (!shared) {
|
|
544
618
|
// Open in read-only mode — MCP server never writes to the database.
|
|
545
619
|
// This allows multiple MCP server instances to read concurrently, and
|
|
@@ -548,7 +622,7 @@ async function doInitLbug(repoId, dbPath) {
|
|
|
548
622
|
for (let attempt = 1; attempt <= LOCK_RETRY_ATTEMPTS; attempt++) {
|
|
549
623
|
try {
|
|
550
624
|
const db = await openReadOnlyDatabase(dbPath);
|
|
551
|
-
shared = { db, refCount: 0, ftsLoaded: false };
|
|
625
|
+
shared = { db, refCount: 0, ftsLoaded: false, dbIdentity: await statDbIdentity(dbPath) };
|
|
552
626
|
dbCache.set(dbPath, shared);
|
|
553
627
|
break;
|
|
554
628
|
}
|
|
@@ -557,7 +631,12 @@ async function doInitLbug(repoId, dbPath) {
|
|
|
557
631
|
if (isWalCorruptionError(lastError)) {
|
|
558
632
|
try {
|
|
559
633
|
const db = await tryQuarantineAndReopen(dbPath, repoId);
|
|
560
|
-
shared = {
|
|
634
|
+
shared = {
|
|
635
|
+
db,
|
|
636
|
+
refCount: 0,
|
|
637
|
+
ftsLoaded: false,
|
|
638
|
+
dbIdentity: await statDbIdentity(dbPath),
|
|
639
|
+
};
|
|
561
640
|
dbCache.set(dbPath, shared);
|
|
562
641
|
break;
|
|
563
642
|
}
|
|
@@ -611,6 +690,9 @@ async function doInitLbug(repoId, dbPath) {
|
|
|
611
690
|
// Register pool entry only after all connections are pre-warmed and FTS is
|
|
612
691
|
// loaded. Concurrent executeQuery calls see either "not initialized"
|
|
613
692
|
// (and throw cleanly) or a fully ready pool — never a half-built one.
|
|
693
|
+
// Record the on-disk identity so a later initLbug can detect an analyze
|
|
694
|
+
// rebuild/mutation and re-open onto the new file (pool staleness invalidation).
|
|
695
|
+
const dbIdentity = await statDbIdentity(dbPath);
|
|
614
696
|
pool.set(repoId, {
|
|
615
697
|
db,
|
|
616
698
|
available,
|
|
@@ -618,6 +700,7 @@ async function doInitLbug(repoId, dbPath) {
|
|
|
618
700
|
waiters: [],
|
|
619
701
|
lastUsed: Date.now(),
|
|
620
702
|
dbPath,
|
|
703
|
+
dbIdentity,
|
|
621
704
|
closed: false,
|
|
622
705
|
});
|
|
623
706
|
ensureIdleTimer();
|
|
@@ -673,6 +756,8 @@ export async function initLbugWithDb(repoId, existingDb, dbPath) {
|
|
|
673
756
|
waiters: [],
|
|
674
757
|
lastUsed: Date.now(),
|
|
675
758
|
dbPath,
|
|
759
|
+
// Injected/external DB (tests) — not tracked for rebuild invalidation.
|
|
760
|
+
dbIdentity: null,
|
|
676
761
|
closed: false,
|
|
677
762
|
});
|
|
678
763
|
ensureIdleTimer();
|
|
@@ -97,6 +97,9 @@ export const runCheckpointWithRetry = async (options = {}) => {
|
|
|
97
97
|
}
|
|
98
98
|
}
|
|
99
99
|
logger.warn({ attempts: CHECKPOINT_RETRY_ATTEMPTS }, 'GitNexus: manual WAL checkpoint exhausted retry budget — surfacing IO error to caller');
|
|
100
|
+
// The held-open cause (#2599) is named at the CLI layer (analyze.ts) where the
|
|
101
|
+
// --wal-checkpoint-threshold recovery hint already renders, so the original IO
|
|
102
|
+
// error is preserved intact for that classifier rather than re-wrapped here.
|
|
100
103
|
throw lastError;
|
|
101
104
|
};
|
|
102
105
|
/**
|
package/dist/core/run-analyze.js
CHANGED
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
*/
|
|
11
11
|
import path from 'path';
|
|
12
12
|
import fs from 'fs/promises';
|
|
13
|
+
import { retryRename } from '../storage/fs-atomic.js';
|
|
13
14
|
import { runPipelineFromRepo } from './ingestion/pipeline.js';
|
|
14
15
|
import { resetDegradedParseCounter } from './tree-sitter/safe-parse.js';
|
|
15
16
|
import { initLbug, loadGraphToLbug, getLbugStats, executeQuery, executeWithReusedStatement, closeLbug, closeLbugBeforeExit, loadCachedEmbeddings, deleteNodesForFiles, deleteAllCommunitiesAndProcesses, deleteAllInterprocTaintPaths, deleteAllCallSummaries, deleteAllInjects, queryImportersBatch, loadFTSExtension, wipeLbugDbFiles, LbugWipeError, DELETE_FILES_CHUNK_SIZE, } from './lbug/lbug-adapter.js';
|
|
@@ -20,7 +21,7 @@ import { cjkSegmentationModeMismatch, getSearchFTSCjkSegmentation, initialiseSea
|
|
|
20
21
|
import { getExtensionCapabilities, resolveAnalyzeInstallPolicy } from './lbug/extension-loader.js';
|
|
21
22
|
import { diagnoseExtensionLoad } from './lbug/extension-load-error.js';
|
|
22
23
|
import { startWalCheckpointDriver, checkpointOnce, } from './lbug/wal-checkpoint-driver.js';
|
|
23
|
-
import { quarantineSidecarsForDirtyRecovery } from './lbug/sidecar-recovery.js';
|
|
24
|
+
import { quarantineSidecarsForDirtyRecovery, inspectLbugSidecars, } from './lbug/sidecar-recovery.js';
|
|
24
25
|
import { getStoragePaths, resolveBranchPlacement, saveMeta, loadMeta, ensureGitNexusIgnored, registerRepo, adoptFlatBranchLabel, isReadOnlyFilesystemError, isRepoRegistered, cleanupOldKuzuFiles, reconcileMetadataFiles, isMissingFilesystemError, INDEX_METADATA_FILE, INCREMENTAL_SCHEMA_VERSION, } from '../storage/repo-manager.js';
|
|
25
26
|
import { DEFAULT_PDG_MAX_FUNCTION_LINES } from './ingestion/cfg/collect.js';
|
|
26
27
|
import { DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, } from './ingestion/cfg/emit.js';
|
|
@@ -901,6 +902,50 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
901
902
|
const hashDiff = isIncremental
|
|
902
903
|
? diffFileHashes(newFileHashes, existingMeta.fileHashes)
|
|
903
904
|
: undefined;
|
|
905
|
+
// #2 atomic index publish: on a full rebuild, build the fresh DB at a temp
|
|
906
|
+
// path and swap it over the live index in one rename at the very end, so a
|
|
907
|
+
// concurrent MCP reader opening mid-build only ever sees the previous
|
|
908
|
+
// complete index (never a wiped/half-built file) and a crash leaves the old
|
|
909
|
+
// index intact. The whole build flows through the singleton connection, so
|
|
910
|
+
// only initLbug/wipeLbugDbFiles below take the temp target.
|
|
911
|
+
//
|
|
912
|
+
// POSIX only: the common CLI/serve-worker analyze paths skip the native close
|
|
913
|
+
// (closeLbugBeforeExit, #2264) and leave the build handle open at swap time.
|
|
914
|
+
// POSIX renames an open file cleanly; a same-process open handle blocks the
|
|
915
|
+
// rename on Windows. Windows keeps the current in-place behavior
|
|
916
|
+
// (buildPath === lbugPath, no swap) until that is resolved (see §12/follow-up).
|
|
917
|
+
const isFullRebuild = !(isIncremental && hashDiff);
|
|
918
|
+
// Where the swap is allowed:
|
|
919
|
+
// - POSIX renames an open file, so the usual skip-native-close (#2264) is
|
|
920
|
+
// fine and the swap always applies.
|
|
921
|
+
// - Windows can swap only when a real close is safe to release the build
|
|
922
|
+
// handle before the rename — i.e. NOT a --pdg run (the #2264 destructor
|
|
923
|
+
// crash). Unverified on Windows CI; falls back to in-place otherwise.
|
|
924
|
+
const posixSwap = process.platform !== 'win32';
|
|
925
|
+
// #2614 Windows: the forced real-close before the rename re-bets that #2264 is
|
|
926
|
+
// --pdg-only, which is unproven (the CLI/worker skip the native close
|
|
927
|
+
// UNCONDITIONALLY) and unverifiable without a Windows runner. Keep it opt-in
|
|
928
|
+
// (GITNEXUS_ATOMIC_WINDOWS_SWAP=1) so the default Windows analyze stays on the
|
|
929
|
+
// proven in-place path; enable it only to test the Windows swap.
|
|
930
|
+
const windowsSwapOk = process.platform === 'win32' &&
|
|
931
|
+
options.pdg !== true &&
|
|
932
|
+
process.env.GITNEXUS_ATOMIC_WINDOWS_SWAP === '1';
|
|
933
|
+
// Incremental atomicity copies the whole index into the temp before mutating
|
|
934
|
+
// it, which negates incremental's speed premise — so it is opt-in
|
|
935
|
+
// (GITNEXUS_ATOMIC_INCREMENTAL=1) pending a benchmark. Full rebuilds always
|
|
936
|
+
// swap where the platform allows.
|
|
937
|
+
const wantAtomicIncremental = isIncremental && !!hashDiff && process.env.GITNEXUS_ATOMIC_INCREMENTAL === '1';
|
|
938
|
+
// #2614 F3: the copy-then-swap stages ONLY the main lbug file, so a live index
|
|
939
|
+
// carrying an orphan .wal/.shadow (a silently-failed prior checkpoint) would
|
|
940
|
+
// be copied incompletely and lose that delta. Only take the atomic path when
|
|
941
|
+
// the live index is a consolidated single file; otherwise fall back to the
|
|
942
|
+
// in-place writeback, which the next open replays correctly.
|
|
943
|
+
const atomicIncremental = wantAtomicIncremental && (await inspectLbugSidecars(lbugPath)).kind === 'clean';
|
|
944
|
+
if (wantAtomicIncremental && !atomicIncremental) {
|
|
945
|
+
log('atomic-incremental: live index carries orphan sidecars — using in-place writeback');
|
|
946
|
+
}
|
|
947
|
+
const useAtomicSwap = (isFullRebuild || atomicIncremental) && (posixSwap || windowsSwapOk);
|
|
948
|
+
const buildPath = useAtomicSwap ? `${lbugPath}.new` : lbugPath;
|
|
904
949
|
if (isIncremental && hashDiff) {
|
|
905
950
|
log(`Incremental: changed=${hashDiff.changed.length}, ` +
|
|
906
951
|
`added=${hashDiff.added.length}, ` +
|
|
@@ -919,6 +964,14 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
919
964
|
directWriteCount: hashDiff.toWrite.length,
|
|
920
965
|
},
|
|
921
966
|
});
|
|
967
|
+
if (atomicIncremental) {
|
|
968
|
+
// Stage the live index into the temp so the in-place delete/writeback
|
|
969
|
+
// below mutates the COPY, and the end-of-run swap publishes it atomically.
|
|
970
|
+
// Clear any stale temp first (a crashed run), then copy the (consolidated,
|
|
971
|
+
// single-file) live index. Whole-file copy — hence opt-in.
|
|
972
|
+
await wipeLbugDbFiles(buildPath);
|
|
973
|
+
await fs.copyFile(lbugPath, buildPath);
|
|
974
|
+
}
|
|
922
975
|
}
|
|
923
976
|
else {
|
|
924
977
|
// Full rebuild path: wipe DB files first.
|
|
@@ -949,7 +1002,12 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
949
1002
|
// valve below can never drift. Failures now throw a typed LbugWipeError
|
|
950
1003
|
// (ENOENT-verified removal) instead of silently letting initLbug reopen
|
|
951
1004
|
// a still-populated DB this run believes it wiped.
|
|
952
|
-
|
|
1005
|
+
//
|
|
1006
|
+
// With the atomic swap (POSIX), this wipes the TEMP build target
|
|
1007
|
+
// (`buildPath` = `<lbugPath>.new`, clearing any stragglers from a crashed
|
|
1008
|
+
// run) and leaves the live index untouched until the end-of-run swap. On
|
|
1009
|
+
// Windows buildPath === lbugPath, so this is the original in-place wipe.
|
|
1010
|
+
await wipeLbugDbFiles(buildPath);
|
|
953
1011
|
}
|
|
954
1012
|
// Size the buffer pool to the graph just built by the pipeline (a page cache
|
|
955
1013
|
// over the on-disk index, which scales with node/edge count) instead of the
|
|
@@ -958,7 +1016,9 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
958
1016
|
// the pool; env override / no-hint paths are unchanged. See
|
|
959
1017
|
// resolveBufferManagerSize / estimateBufferPool.
|
|
960
1018
|
setBufferPoolSizeHint(estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount));
|
|
961
|
-
|
|
1019
|
+
// Full rebuild (POSIX) builds into the temp `buildPath`; incremental and
|
|
1020
|
+
// Windows use `buildPath === lbugPath` in place.
|
|
1021
|
+
await initLbug(buildPath);
|
|
962
1022
|
// Manual WAL checkpoint driver (#1741): periodically drain the WAL
|
|
963
1023
|
// from JS so the un-retriable native auto-checkpoint almost never
|
|
964
1024
|
// has work left to do. Failures of the manual CHECKPOINT are absorbed
|
|
@@ -1191,8 +1251,8 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
1191
1251
|
// to replace wholesale.
|
|
1192
1252
|
await walCheckpointDriver.stop();
|
|
1193
1253
|
await closeLbug();
|
|
1194
|
-
await wipeLbugDbFiles(
|
|
1195
|
-
await initLbug(
|
|
1254
|
+
await wipeLbugDbFiles(buildPath);
|
|
1255
|
+
await initLbug(buildPath);
|
|
1196
1256
|
walCheckpointDriver = startWalCheckpointDriver();
|
|
1197
1257
|
await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
|
|
1198
1258
|
lbugMsgCount++;
|
|
@@ -1726,7 +1786,11 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
1726
1786
|
// inside the resolver, and a mismatch leaves the dirty flag intact so the
|
|
1727
1787
|
// next run takes the established full-recovery path.
|
|
1728
1788
|
meta.runnerIdentity = finalizeAnalyzerRunnerIdentity(import.meta.url, runnerIdentity);
|
|
1729
|
-
|
|
1789
|
+
// #2614 F1: the freshness stamp (saveMeta) is written AFTER the atomic swap
|
|
1790
|
+
// below — never here — so a concurrent MCP reader can't observe
|
|
1791
|
+
// meta.indexedAt = T_new while lbugPath still resolves to the pre-swap
|
|
1792
|
+
// inode (which latched the reader on the stale index permanently). The meta
|
|
1793
|
+
// object is fully computed at this point; only its write is deferred.
|
|
1730
1794
|
// Persist the incremental parse cache for the next run. Wraps in
|
|
1731
1795
|
// try/catch so a cache-write failure never breaks an otherwise
|
|
1732
1796
|
// successful indexing run. Prune stale chunk-hash entries first so
|
|
@@ -1850,7 +1914,54 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
|
|
|
1850
1914
|
// LadybugDB destructor double-free after --pdg writes — closeLbugBeforeExit
|
|
1851
1915
|
// CHECKPOINTs for durability then leaves the handles for process exit to
|
|
1852
1916
|
// reclaim (#2264). Long-lived callers close for real.
|
|
1853
|
-
|
|
1917
|
+
//
|
|
1918
|
+
// On Windows a swap must release the build handle before the rename (a
|
|
1919
|
+
// same-process open file can't be renamed), so it forces a real close —
|
|
1920
|
+
// safe because windowsSwapOk excludes --pdg (the #2264 case). POSIX renames
|
|
1921
|
+
// an open file, so it keeps the skip-native-close there.
|
|
1922
|
+
const forceRealCloseForSwap = useAtomicSwap && process.platform === 'win32';
|
|
1923
|
+
await (options.skipNativeCloseOnExit && !forceRealCloseForSwap
|
|
1924
|
+
? closeLbugBeforeExit()
|
|
1925
|
+
: closeLbug());
|
|
1926
|
+
// #2 atomic publish: the fresh index was built at buildPath (a full rebuild,
|
|
1927
|
+
// or an opt-in atomic incremental that copied the live index in first). Swap
|
|
1928
|
+
// it over the live lbugPath in one rename so an MCP reader that opened
|
|
1929
|
+
// mid-build only ever saw the previous complete index — never a wiped/
|
|
1930
|
+
// half-built file. The close above checkpoint-consolidated buildPath to a
|
|
1931
|
+
// single file (no .wal), so the rename publishes a complete index; a reader
|
|
1932
|
+
// holding the old inode keeps a consistent stale snapshot until the pool
|
|
1933
|
+
// re-opens onto the new one (the pool staleness invalidation). Runs only on
|
|
1934
|
+
// success — a thrown error skips this, leaving the live index intact and the
|
|
1935
|
+
// temp build to be cleared by the next run's wipe.
|
|
1936
|
+
// Only publish if the build actually produced a DB at buildPath. A
|
|
1937
|
+
// degenerate run (empty repo, or a mocked pipeline that never opened the
|
|
1938
|
+
// store) leaves nothing to swap — skip rather than throw ENOENT.
|
|
1939
|
+
const builtDbExists = useAtomicSwap
|
|
1940
|
+
? await fs.stat(buildPath).then(() => true, () => false)
|
|
1941
|
+
: false;
|
|
1942
|
+
if (useAtomicSwap && builtDbExists) {
|
|
1943
|
+
await retryRename(buildPath, lbugPath);
|
|
1944
|
+
// Clear any sidecars orphaned beside the replaced file. A cleanly-closed
|
|
1945
|
+
// prior index has none; a crashed one could, and it would be replay
|
|
1946
|
+
// poison next to the freshly published index. Best-effort.
|
|
1947
|
+
for (const suffix of ['.wal', '.shadow', '.wal.checkpoint']) {
|
|
1948
|
+
await fs.rm(`${lbugPath}${suffix}`, { force: true }).catch(() => { });
|
|
1949
|
+
}
|
|
1950
|
+
// #2614 F4: if the final checkpoint silently failed, the build may still
|
|
1951
|
+
// carry a residual .wal/.shadow under the temp name. MOVE it beside the
|
|
1952
|
+
// published index (not orphan/delete it) so the next open replays the
|
|
1953
|
+
// delta, rather than leaving it under a name LadybugDB never reconciles.
|
|
1954
|
+
for (const suffix of ['.wal', '.shadow']) {
|
|
1955
|
+
await fs.rename(`${buildPath}${suffix}`, `${lbugPath}${suffix}`).catch(() => { });
|
|
1956
|
+
}
|
|
1957
|
+
}
|
|
1958
|
+
// #2614 F1: stamp the freshness metadata now that the index is published.
|
|
1959
|
+
// When meta.indexedAt becomes visible, lbugPath already resolves to the new
|
|
1960
|
+
// inode, so a reader reiniting on the stamp opens the fresh graph rather
|
|
1961
|
+
// than latching on the old one. Leaving the dirty flag set across the swap
|
|
1962
|
+
// is a crash-safety improvement: a failed swap leaves the previous index
|
|
1963
|
+
// live and the next run recovers via the full-rebuild path.
|
|
1964
|
+
await saveMeta(metaDir, meta);
|
|
1854
1965
|
progress('done', 100, 'Done');
|
|
1855
1966
|
return {
|
|
1856
1967
|
repoName: projectName,
|
|
@@ -147,6 +147,7 @@ export declare class LocalBackend {
|
|
|
147
147
|
private reinitPromises;
|
|
148
148
|
private lastStalenessCheck;
|
|
149
149
|
private lastObservedIndexedAt;
|
|
150
|
+
private lastObservedDbIdentity;
|
|
150
151
|
private groupToolSvc;
|
|
151
152
|
/**
|
|
152
153
|
* One-shot stderr warnings for sibling-clone drift, keyed by
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
import fs from 'fs/promises';
|
|
9
9
|
import path from 'path';
|
|
10
10
|
import { createHash } from 'crypto';
|
|
11
|
-
import { initLbug, executeQuery, executeParameterized, closeLbug, isLbugReady, } from '../../core/lbug/pool-adapter.js';
|
|
11
|
+
import { initLbug, executeQuery, executeParameterized, closeLbug, isLbugReady, statDbIdentity, dbIdentityChanged, } from '../../core/lbug/pool-adapter.js';
|
|
12
12
|
import { queryClassBeanMetadata } from './bean-metadata.js';
|
|
13
13
|
import { isValidQueryParams } from '../../core/lbug/query-params.js';
|
|
14
14
|
import { toDisplayLine } from './line-display.js';
|
|
@@ -470,6 +470,11 @@ export class LocalBackend {
|
|
|
470
470
|
// not persist across calls and the staleness check would reinit forever
|
|
471
471
|
// (#2106).
|
|
472
472
|
lastObservedIndexedAt = new Map();
|
|
473
|
+
// #2614 F1: file identity of the lbug the pool last opened. An atomic swap or
|
|
474
|
+
// an in-place incremental changes the inode; reiniting on that reinit-covers
|
|
475
|
+
// the window where meta.indexedAt hasn't caught up (and the incremental case),
|
|
476
|
+
// so a rebuilt index is never served stale even when the stamp looks current.
|
|
477
|
+
lastObservedDbIdentity = new Map();
|
|
473
478
|
groupToolSvc = null;
|
|
474
479
|
/**
|
|
475
480
|
* One-shot stderr warnings for sibling-clone drift, keyed by
|
|
@@ -755,6 +760,7 @@ export class LocalBackend {
|
|
|
755
760
|
this.initializedRepos.delete(key);
|
|
756
761
|
this.lastStalenessCheck.delete(key);
|
|
757
762
|
this.lastObservedIndexedAt.delete(key);
|
|
763
|
+
this.lastObservedDbIdentity.delete(key);
|
|
758
764
|
this.reinitPromises.delete(key);
|
|
759
765
|
closeLbug(key).catch(() => { });
|
|
760
766
|
}
|
|
@@ -1114,23 +1120,38 @@ export class LocalBackend {
|
|
|
1114
1120
|
// Reading the flat meta for a branch handle would compare the branch
|
|
1115
1121
|
// index's indexedAt against the primary's and thrash the pool (#2106).
|
|
1116
1122
|
const meta = await loadMeta(path.dirname(repo.lbugPath));
|
|
1117
|
-
if (!meta)
|
|
1118
|
-
return;
|
|
1119
1123
|
// Compare against the last indexedAt OBSERVED for this pool (keyed by
|
|
1120
1124
|
// lbugPath), not the handle's — branch handles are fresh spreads so a
|
|
1121
1125
|
// handle mutation would not persist and would reinit on every check.
|
|
1122
1126
|
const observed = this.lastObservedIndexedAt.get(poolKey) ?? repo.indexedAt;
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
+
const stampChanged = !!meta?.indexedAt && meta.indexedAt !== observed;
|
|
1128
|
+
// #2614 F1: also reinit on a file-identity change. An atomic swap (or an
|
|
1129
|
+
// in-place incremental) changes the lbug inode; keying only on
|
|
1130
|
+
// meta.indexedAt let a reader that reinited inside the pre-swap window
|
|
1131
|
+
// latch on the old inode forever (its stamp already == meta.indexedAt).
|
|
1132
|
+
const currentIdentity = await statDbIdentity(repo.lbugPath);
|
|
1133
|
+
const identityChanged = dbIdentityChanged(this.lastObservedDbIdentity.get(poolKey) ?? null, currentIdentity);
|
|
1134
|
+
if (stampChanged || identityChanged) {
|
|
1135
|
+
// Index was rebuilt/swapped — DELEGATE the close/reopen to the pool's
|
|
1136
|
+
// initLbug, which refuses to evict (and close the shared Database)
|
|
1137
|
+
// while a query is in flight (its checkedOut>0 guard). Calling
|
|
1138
|
+
// closeLbug directly here bypassed that guard and could close a
|
|
1139
|
+
// Database mid-query — a native use-after-free (#2614). Wrap in
|
|
1140
|
+
// reinitPromises to serialize concurrent detectors.
|
|
1127
1141
|
const reinit = (async () => {
|
|
1128
1142
|
try {
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1143
|
+
// Advance the observed stamp regardless: a stamp change with an
|
|
1144
|
+
// unchanged file must not re-trigger on every check.
|
|
1145
|
+
if (meta?.indexedAt)
|
|
1146
|
+
this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
|
|
1147
|
+
const reopened = await initLbug(poolKey, repo.lbugPath);
|
|
1148
|
+
// Advance the observed IDENTITY only when the pool actually rolled
|
|
1149
|
+
// over. If a query was in flight, initLbug served the current
|
|
1150
|
+
// handle and returned false; leaving the identity divergent
|
|
1151
|
+
// re-triggers the reopen on a later idle check instead of latching.
|
|
1152
|
+
if (reopened) {
|
|
1153
|
+
this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
|
|
1154
|
+
}
|
|
1134
1155
|
}
|
|
1135
1156
|
finally {
|
|
1136
1157
|
this.reinitPromises.delete(poolKey);
|
|
@@ -1151,6 +1172,7 @@ export class LocalBackend {
|
|
|
1151
1172
|
await initLbug(poolKey, repo.lbugPath);
|
|
1152
1173
|
this.initializedRepos.add(poolKey);
|
|
1153
1174
|
this.lastObservedIndexedAt.set(poolKey, repo.indexedAt);
|
|
1175
|
+
this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
|
|
1154
1176
|
}
|
|
1155
1177
|
catch (err) {
|
|
1156
1178
|
// If lock error, mark as not initialized so next call retries
|
package/package.json
CHANGED