gitnexus 1.6.12-rc.38 → 1.6.12-rc.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,6 +4,7 @@ import fs from 'node:fs/promises';
4
4
  import { watch } from 'chokidar';
5
5
  import { createWatchIgnorePredicate } from '../config/ignore-service.js';
6
6
  import { analyzeFailureMayHaveMutatedLiveIndex, runFullAnalysis, } from '../core/run-analyze.js';
7
+ import { isIndexLockGuardTimeout } from '../storage/index-lock.js';
7
8
  import { getGitRoot, hasGitDir } from '../storage/git.js';
8
9
  import { GITNEXUS_DIR } from '../storage/repo-meta.js';
9
10
  import { loadAnalyzeConfigStrict, mergeAnalyzeOptions, validateBranchName, } from './analyze-config.js';
@@ -173,6 +174,9 @@ class WatchControlReloadError extends Error {
173
174
  }
174
175
  }
175
176
  export function shouldStopAfterWatchRefreshFailure(error, paths) {
177
+ if (isIndexLockGuardTimeout(error)) {
178
+ return true;
179
+ }
176
180
  return (paths.length > 0 &&
177
181
  !(error instanceof WatchControlReloadError) &&
178
182
  analyzeFailureMayHaveMutatedLiveIndex(error));
@@ -397,8 +401,13 @@ export async function watchCommandWithRunnerIdentity(runnerIdentityAtBootstrap,
397
401
  const detail = paths.length > 0 ? ` (${paths.length} queued path(s))` : '';
398
402
  if (shouldStopAfterWatchRefreshFailure(error, paths)) {
399
403
  fatalRefreshError = error;
404
+ const guardTimeout = isIndexLockGuardTimeout(error);
400
405
  cliError(`Refresh failed${detail}: ${error instanceof Error ? error.message : String(error)}. ` +
401
- 'Watch mode is stopping because the live index may have been updated in place.');
406
+ (guardTimeout
407
+ ? 'Watch mode is stopping because the acquisition guard needs quiesced recovery; see RUNBOOK.md.'
408
+ : 'Watch mode is stopping because the live index may have been updated in place.'), guardTimeout
409
+ ? { recoveryHint: 'index-lock-guard-recovery', guardPath: error.guardPath }
410
+ : undefined);
402
411
  stopWatching();
403
412
  return;
404
413
  }
@@ -20,7 +20,7 @@ import { causeChain } from '../lib/utils.js';
20
20
  import { getOsPageSize, isLbugCheckpointIoError, isLbugCheckpointBusyError, isLbugPageSizeFrameError, isPageSizeAwareLadybug, isWalCorruptionError, parseWalCheckpointThreshold, WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js';
21
21
  import { getStoragePaths, getGlobalRegistryPath, RegistryNameCollisionError, AnalysisNotFinalizedError, assertAnalysisFinalized, } from '../storage/repo-manager.js';
22
22
  import { getGitRoot, hasGitDir, getDefaultBranch, selfCommitContextFiles, snapshotSelfCommitSafety, } from '../storage/git.js';
23
- import { IndexLockTimeoutError } from '../storage/index-lock.js';
23
+ import { IndexLockTimeoutError, isIndexLockGuardTimeout } from '../storage/index-lock.js';
24
24
  import { loadAnalyzeConfig, mergeAnalyzeOptions, resolveDefaultBranch, validateBranchName, GitNexusRcError, } from './analyze-config.js';
25
25
  import { runFullAnalysis } from '../core/run-analyze.js';
26
26
  import { getRuntimeFingerprint } from '../core/platform/capabilities.js';
@@ -1377,6 +1377,14 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
1377
1377
  // refreshed by the holder — this is a clean, expected condition, not a
1378
1378
  // crash, so render the message without a stack trace.
1379
1379
  if (err instanceof IndexLockTimeoutError) {
1380
+ if (isIndexLockGuardTimeout(err)) {
1381
+ cliError(err.message, {
1382
+ recoveryHint: 'index-lock-guard-recovery',
1383
+ guardPath: err.guardPath,
1384
+ });
1385
+ process.exitCode = 1;
1386
+ return;
1387
+ }
1380
1388
  cliError(` Another gitnexus analyze (pid ${err.holder.pid} on ${err.holder.hostname}) is ` +
1381
1389
  `already refreshing this index and did not finish within the wait window.\n` +
1382
1390
  ` The on-disk index is being updated by that run. Retry later, or raise\n` +
@@ -12,7 +12,7 @@ import { type CliMessageKey, type CliMessageVars } from './i18n/index.js';
12
12
  * Consumers can import this type to narrow log-record `recoveryHint`
13
13
  * fields without restating the literal list.
14
14
  */
15
- export type RecoveryHint = 'wal-corruption' | 'wal-checkpoint-threshold' | 'lbug-wipe-failed' | 'lbug-page-size' | 'heap-oom-respawn' | 'native-worker-abort' | 'hf-endpoint-unreachable' | 'http-embedding-endpoint-error' | 'embedding-dims-invalid' | 'local-embedding-unsupported' | 'local-embedding-stack-missing' | 'large-repo' | 'npm-resolution' | 'module-not-found' | 'gitnexusrc-invalid' | 'default-branch-invalid' | 'index-lock-timeout' | 'undeclared-relation-pair';
15
+ export type RecoveryHint = 'wal-corruption' | 'wal-checkpoint-threshold' | 'lbug-wipe-failed' | 'lbug-page-size' | 'heap-oom-respawn' | 'native-worker-abort' | 'hf-endpoint-unreachable' | 'http-embedding-endpoint-error' | 'embedding-dims-invalid' | 'local-embedding-unsupported' | 'local-embedding-stack-missing' | 'large-repo' | 'npm-resolution' | 'module-not-found' | 'gitnexusrc-invalid' | 'default-branch-invalid' | 'index-lock-timeout' | 'index-lock-guard-recovery' | 'undeclared-relation-pair';
16
16
  /**
17
17
  * Common shape for the optional structured-field bag passed to
18
18
  * `cliError`/`cliWarn`/`cliInfo`. Typed so the `recoveryHint` slot is
@@ -2,7 +2,7 @@ import { lstat } from 'node:fs/promises';
2
2
  import path from 'node:path';
3
3
  import { cliInfo } from './cli-message.js';
4
4
  import { getGitRoot } from '../storage/git.js';
5
- import { acquireIndexLock } from '../storage/index-lock.js';
5
+ import { acquireIndexLock, requireExclusiveIndexLock } from '../storage/index-lock.js';
6
6
  import { getStoragePaths, loadMeta, saveMeta } from '../storage/repo-manager.js';
7
7
  import { closeLbug, executeQuery, executeWithReusedStatement, fetchExistingEmbeddingHashes, initLbug, } from '../core/lbug/lbug-adapter.js';
8
8
  import { runEmbeddingPipeline } from '../core/embeddings/embedding-pipeline.js';
@@ -19,6 +19,7 @@ export const embeddingsSyncCommand = async (inputPath) => {
19
19
  const metaDir = path.dirname(metaPath);
20
20
  const lock = await acquireIndexLock(metaDir);
21
21
  try {
22
+ requireExclusiveIndexLock(lock, `Cannot acquire the index lock at ${metaDir}; refusing an unlocked embeddings sync.`);
22
23
  const meta = await loadMeta(metaDir);
23
24
  if (!meta)
24
25
  throw new Error(`No GitNexus index found for ${repoPath}. Run gitnexus analyze first.`);
@@ -30,7 +30,9 @@
30
30
  * distinct ways it can fail to be protected; all three throw
31
31
  * {@link GroupSyncLockError}:
32
32
  *
33
- * 1. TIMEOUT — the holder is still alive when the ceiling elapses.
33
+ * 1. TIMEOUT — a live holder still held the lock when the ceiling
34
+ * elapsed, or an unrecoverable acquisition/reclaim guard leftover
35
+ * exhausted the guard wait (see RUNBOOK.md).
34
36
  * 2. LOCK-FREE DEGRADATION — `acquireIndexLock` answers a read-only or
35
37
  * permission-denied filesystem with a no-op handle that is byte-identical
36
38
  * to a real one at the API boundary. That is a deliberate tolerance for
@@ -69,7 +71,7 @@
69
71
  * `GITNEXUS_INDEX_LOCK_BACKEND=file` is what covers that deployment.
70
72
  */
71
73
  import path from 'node:path';
72
- import { acquireIndexLock, IndexLockTimeoutError, } from '../../storage/index-lock.js';
74
+ import { acquireIndexLock, IndexLockTimeoutError, isIndexLockGuardTimeout, } from '../../storage/index-lock.js';
73
75
  import { logger } from '../logger.js';
74
76
  /** Lock-directory name inside the group directory. Never the group dir itself. */
75
77
  export const GROUP_SYNC_LOCK_DIRNAME = 'sync-lock';
@@ -107,7 +109,7 @@ export class GroupSyncLockError extends Error {
107
109
  export const withGroupSyncLock = async (groupDir, operation) => {
108
110
  let handle;
109
111
  // The wrapper times the acquisition itself. `IndexLockTimeoutError` carries
110
- // `holder` and `holderKnown` and nothing else — the elapsed wait exists only
112
+ // holder/guard identity but no elapsed-time field — the elapsed wait exists only
111
113
  // inside its inherited message string, so the figure has to be measured here
112
114
  // to be reported without that message. `Date.now()` matches how the primitive
113
115
  // measures its own wait.
@@ -128,6 +130,11 @@ export const withGroupSyncLock = async (groupDir, operation) => {
128
130
  // socket backend the holder is not identifiable at all. Re-word it around
129
131
  // what IS known: which group, which operation, and how long we waited.
130
132
  if (err instanceof IndexLockTimeoutError) {
133
+ if (isIndexLockGuardTimeout(err)) {
134
+ throw new GroupSyncLockError('timeout', groupDir, `Could not acquire the sync lock for group "${path.basename(groupDir)}" ` +
135
+ `(${getGroupSyncLockDir(groupDir)}). ${err.message} ` +
136
+ `Nothing was written and this group was not synced.`, err);
137
+ }
131
138
  throw new GroupSyncLockError('timeout', groupDir, `Timed out after ${Date.now() - acquireStartedAt}ms waiting for the sync lock on ` +
132
139
  `group "${path.basename(groupDir)}" (${getGroupSyncLockDir(groupDir)}). ` +
133
140
  // `holderKnown` is false on the socket backend and on the file
@@ -16,7 +16,7 @@ import fs from 'fs/promises';
16
16
  import { constants as fsConstants } from 'node:fs';
17
17
  import { randomUUID } from 'node:crypto';
18
18
  import { retryRename } from '../storage/fs-atomic.js';
19
- import { acquireIndexLock } from '../storage/index-lock.js';
19
+ import { acquireIndexLock, requireExclusiveIndexLock } from '../storage/index-lock.js';
20
20
  import { invalidateNodeWorkspacePackages } from './ingestion/import-resolvers/node-workspace-packages.js';
21
21
  import { logNameFallbackSummary, summarizeNameFallback, countCallsByLanguage, } from './ingestion/scope-resolution/name-fallback-summary.js';
22
22
  import { runPipelineFromRepo } from './ingestion/pipeline.js';
@@ -609,6 +609,7 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
609
609
  let writeTarget = await resolveWriteTarget(repoPath, options);
610
610
  let lock = await acquireIndexLock(writeTarget.metaDir, acquireOpts);
611
611
  try {
612
+ requireExclusiveIndexLock(lock, `Cannot acquire the index lock at ${writeTarget.metaDir}; refusing an unlocked analysis.`);
612
613
  // #2658 review H2: acquireIndexLock can wait up to the timeout ceiling,
613
614
  // during which git HEAD/branch — and thus the resolved write slot — may
614
615
  // change (a commit lands, a branch is switched, or another writer adopts the
@@ -633,6 +634,7 @@ export async function runFullAnalysis(repoPath, options, callbacks, runnerIdenti
633
634
  lock.release();
634
635
  writeTarget = fresh;
635
636
  lock = await acquireIndexLock(fresh.metaDir, acquireOpts);
637
+ requireExclusiveIndexLock(lock, `Cannot acquire the index lock at ${fresh.metaDir}; refusing an unlocked analysis.`);
636
638
  if (attempt === MAX_RELOCK - 1) {
637
639
  log('Index write target still moving after repeated re-acquire; proceeding on this lock.');
638
640
  }
@@ -1,7 +1,7 @@
1
1
  import { projectAnalyzeResultForIpc } from './analyze-worker-ipc.js';
2
2
  // Value import (instanceof): index-lock is a lightweight storage primitive
3
3
  // (node:fs/net/crypto only), so this does NOT pull in run-analyze/repo-manager.
4
- import { IndexLockTimeoutError } from '../storage/index-lock.js';
4
+ import { IndexLockTimeoutError, isIndexLockGuardTimeout } from '../storage/index-lock.js';
5
5
  /**
6
6
  * Run the analysis and report the outcome to the parent over IPC. Reports at most
7
7
  * one terminal message (`complete` or `error`) — and none if a cancellation
@@ -39,9 +39,15 @@ export async function runWorkerAnalysis(repoPath, options, deps, runnerIdentityA
39
39
  // #2658 review M2: a lock-wait timeout is transient contention (another
40
40
  // analyze held the single-writer lock), not a broken build — tag it so the
41
41
  // parent can surface a retry signal instead of an opaque hard failure.
42
+ // An orphan guard needs quiesced recovery, not automatic retries.
42
43
  terminal =
43
44
  err instanceof IndexLockTimeoutError
44
- ? { type: 'error', message, code: 'index-lock-timeout', retryable: true }
45
+ ? {
46
+ type: 'error',
47
+ message,
48
+ code: 'index-lock-timeout',
49
+ retryable: !isIndexLockGuardTimeout(err),
50
+ }
45
51
  : { type: 'error', message };
46
52
  }
47
53
  // P3 (#2264): only report if a SIGTERM cancellation hasn't already claimed the
@@ -52,12 +52,17 @@ export interface ErrorMessage {
52
52
  /**
53
53
  * Machine-readable failure code for a parent that wants to branch instead of
54
54
  * only surfacing the string. `index-lock-timeout` (#2658 review M2) means
55
- * another analyze held the single-writer lock past the wait ceiling — a
56
- * transient, retryable condition, not a broken build. Absent for a generic
57
- * failure.
55
+ * acquisition waited past a wait ceiling. Ordinary lock contention is
56
+ * transient (`retryable: true`). An orphan acquisition/reclaim guard
57
+ * (`IndexLockTimeoutError.guardPath` set) is not — retries re-hit the 30s
58
+ * cap and need quiesced recovery (RUNBOOK.md), so `retryable` is false.
59
+ * Absent for a generic failure.
58
60
  */
59
61
  code?: 'index-lock-timeout';
60
- /** True when the failure is expected to clear on retry (e.g. lock contention). */
62
+ /**
63
+ * True when the failure is expected to clear on retry (e.g. lock contention).
64
+ * False when `code` is `index-lock-timeout` because of guard contention.
65
+ */
61
66
  retryable?: boolean;
62
67
  }
63
68
  /** Child → parent IPC messages. Shared with the parent-side launcher. */
@@ -1,8 +1,9 @@
1
1
  declare const LOCK_RECORD_VERSION: 1;
2
2
  /**
3
- * On-disk lock record. `token` proves ownership on release/steal; `startTime`
4
- * (Linux only) defends against pid reuse; `invocationId` is a human-traceable
5
- * id distinct from the security-irrelevant `token`.
3
+ * On-disk lock record. `token` proves ownership on release and guard/workload
4
+ * verification; `startTime` (Linux only) defends against pid reuse;
5
+ * `invocationId` is a human-traceable id distinct from the security-irrelevant
6
+ * `token`.
6
7
  */
7
8
  export interface LockRecord {
8
9
  v: typeof LOCK_RECORD_VERSION;
@@ -50,7 +51,9 @@ export interface AcquireOptions {
50
51
  * would otherwise block acquisition forever. Timing out is safe — it stops
51
52
  * *waiting*, never *steals* a possibly-live holder — and names the holder so
52
53
  * the caller can retry. Override (including to unbounded, value ≤ 0) via
53
- * GITNEXUS_INDEX_LOCK_TIMEOUT_MS.
54
+ * GITNEXUS_INDEX_LOCK_TIMEOUT_MS. File acquisition/reclaim guard contention
55
+ * is separately capped at 30s, or the remaining timeout if shorter; an
56
+ * orphan guard is never automatically taken over.
54
57
  */
55
58
  timeoutMs?: number;
56
59
  /** Base poll interval (ms); jittered. Default 250. */
@@ -68,15 +71,19 @@ export declare class IndexLockTimeoutError extends Error {
68
71
  * pid when this is false.
69
72
  */
70
73
  readonly holderKnown: boolean;
71
- constructor(holder: LockRecord, waitedMs: number, holderKnown?: boolean);
74
+ /** Present only for acquisition/reclaim guard contention, requiring quiesced recovery. */
75
+ readonly guardPath?: string;
76
+ constructor(holder: LockRecord, waitedMs: number, holderKnown?: boolean, guardPath?: string);
72
77
  }
78
+ export declare const isIndexLockGuardTimeout: (error: unknown) => error is IndexLockTimeoutError & {
79
+ guardPath: string;
80
+ };
81
+ /** Writers must refuse a handle that does not own the lock. */
82
+ export declare const requireExclusiveIndexLock: (handle: IndexLockHandle, message: string) => void;
73
83
  /**
74
- * Filesystem-create error codes we tolerate by proceeding lock-free: a
75
- * read-only mount (EROFS) or a denied create (EACCES/EPERM). Such a filesystem
76
- * rejects every index WRITE in the same directory too, so no concurrent writer
77
- * can exist and the lock is moot — an already-indexed repo on a `:ro` mount
78
- * must still reach its `alreadyUpToDate` fast path (#2658). A genuinely-needed
79
- * write fails later exactly as it would have without the lock.
84
+ * Filesystem-create errors eligible for a read-only, non-owning handle when
85
+ * neither lock nor guard exists. Denied creation does NOT prove other writers
86
+ * lack access (ACLs may differ). Callers that write must reject lockFree handles.
80
87
  */
81
88
  export declare const LOCK_UNWRITABLE_CODES: ReadonlySet<string>;
82
89
  export declare const isLockUnwritableCode: (code: string | undefined) => boolean;
@@ -33,25 +33,28 @@
33
33
  * - **file** (`O_EXCL` pidfile) — the portable fallback for macOS/BSD (no
34
34
  * abstract sockets; filesystem sockets don't release cleanly on death) and
35
35
  * for any environment where the socket backend can't bind. It uses pid-
36
- * liveness staleness, an atomic rename-steal reclaim, bounded malformed-file
37
- * handling, read-only tolerance, and a finite wait timeout (a reused pid can
38
- * masquerade as live where process start-time isn't verifiable, so waiting is
39
- * bounded rather than a hang). Its stale-takeover has an irreducible narrow
40
- * race — inherent to file-based advisory locks — which is precisely why the
41
- * socket backend is preferred; only a kernel primitive closes it.
36
+ * liveness staleness, a non-stealable acquisition/reclaim guard, bounded
37
+ * malformed-file handling, read-only tolerance, and a finite wait timeout (a
38
+ * reused pid can masquerade as live where process start-time isn't verifiable, so waiting is
39
+ * bounded rather than a hang). Every file acquisition participates in the
40
+ * guard, even for an empty slot. An orphan guard fails closed and requires
41
+ * quiesced manual recovery (RUNBOOK.md); socket locks avoid that tradeoff.
42
+ * Requires reliable local-filesystem O_EXCL and cooperating upgraded writers.
42
43
  *
43
44
  * Scope: cross-process, same logical index dir. The file backend never steals a
44
45
  * foreign-host lock (pid liveness is meaningless across hosts); the socket
45
46
  * backend is single-host by nature. The motivating case (local hook-driven
46
47
  * re-index) is single-host. See AcquireOptions.timeoutMs for the wait ceiling.
47
48
  */
48
- import { openSync, writeSync, closeSync, readFileSync, unlinkSync, renameSync, existsSync, mkdirSync, readdirSync, realpathSync, } from 'node:fs';
49
+ import { openSync, writeSync, closeSync, readFileSync, unlinkSync, mkdirSync, readdirSync, realpathSync, lstatSync, } from 'node:fs';
49
50
  import net from 'node:net';
50
51
  import path from 'node:path';
51
52
  import os from 'node:os';
52
53
  import { randomBytes, randomUUID, createHash } from 'node:crypto';
54
+ import { isProcessAlive } from '../utils/process-identity.js';
53
55
  const LOCK_FILENAME = 'analyze.lock';
54
56
  const LOCK_RECORD_VERSION = 1;
57
+ const lockGuardPath = (lockPath) => `${lockPath}.guard`;
55
58
  /** Base poll interval while waiting for a live holder; jittered per attempt. */
56
59
  const DEFAULT_POLL_MS = 250;
57
60
  /** How often to re-emit the "still waiting for pid N" diagnostic. */
@@ -64,12 +67,13 @@ const DIAGNOSTIC_INTERVAL_MS = 15_000;
64
67
  * GITNEXUS_INDEX_LOCK_TIMEOUT_MS (or set it ≤ 0 for unbounded).
65
68
  */
66
69
  const DEFAULT_TIMEOUT_MS = 600_000;
70
+ /** Never wait indefinitely for a guard, even with an unbounded workload wait. */
71
+ const GUARD_TIMEOUT_MS = 30_000;
67
72
  /**
68
73
  * How long a lock file must stay unreadable (empty/partial JSON) before we
69
- * treat it as a crash orphan and reclaim it. Tolerates the microsecond
70
- * create→write→close window of a *live* owner (see acquireIndexLock), so we
71
- * never steal a lock that is a poll-interval away from being written. Scaled
72
- * off the poll interval, floored at 1s.
74
+ * treat it as a crash orphan and reclaim it. This grace preserves legacy
75
+ * orphan handling; the guard, NOT elapsed time, excludes incomplete live
76
+ * creations. Scaled off the poll interval, floored at 1s.
73
77
  */
74
78
  const malformedGraceMs = (pollMs) => Math.max(1000, pollMs * 2);
75
79
  export class IndexLockTimeoutError extends Error {
@@ -82,18 +86,38 @@ export class IndexLockTimeoutError extends Error {
82
86
  * pid when this is false.
83
87
  */
84
88
  holderKnown;
85
- constructor(holder, waitedMs, holderKnown = true) {
86
- super(holderKnown
87
- ? `Timed out after ${waitedMs}ms waiting for another gitnexus analyze ` +
88
- `(pid ${holder.pid} on ${holder.hostname}, invocation ${holder.invocationId}) ` +
89
- `to release the index lock.`
90
- : `Timed out after ${waitedMs}ms waiting for another gitnexus analyze ` +
91
- `(holder identity unknown) to release the index lock.`);
89
+ /** Present only for acquisition/reclaim guard contention, requiring quiesced recovery. */
90
+ guardPath;
91
+ constructor(holder, waitedMs, holderKnown = true, guardPath) {
92
+ super(formatIndexLockTimeoutMessage(holder, waitedMs, holderKnown, guardPath));
92
93
  this.name = 'IndexLockTimeoutError';
93
94
  this.holder = holder;
94
95
  this.holderKnown = holderKnown;
96
+ this.guardPath = guardPath;
95
97
  }
96
98
  }
99
+ export const isIndexLockGuardTimeout = (error) => error instanceof IndexLockTimeoutError && error.guardPath !== undefined;
100
+ /** Writers must refuse a handle that does not own the lock. */
101
+ export const requireExclusiveIndexLock = (handle, message) => {
102
+ if (handle.lockFree)
103
+ throw new Error(message);
104
+ };
105
+ const formatIndexLockTimeoutMessage = (holder, waitedMs, holderKnown, guardPath) => {
106
+ if (guardPath !== undefined) {
107
+ return (`Timed out after ${waitedMs}ms waiting for acquisition/reclaim guard ${guardPath}. ` +
108
+ `Quiesce all relevant writers and prevent restart before manual recovery. ` +
109
+ `Never remove the guard while writers may run; see RUNBOOK.md for quiesced recovery.`);
110
+ }
111
+ if (holderKnown) {
112
+ return (`Timed out after ${waitedMs}ms waiting for another gitnexus analyze ` +
113
+ `(pid ${holder.pid} on ${holder.hostname}, invocation ${holder.invocationId}) ` +
114
+ `to release the index lock.`);
115
+ }
116
+ return (`Timed out after ${waitedMs}ms waiting for another gitnexus analyze ` +
117
+ `(holder identity unknown) to release the index lock.`);
118
+ };
119
+ const unverifiedGuardError = (guardPath) => new Error(`Cannot verify acquisition/reclaim guard ownership: ${guardPath}. ` +
120
+ 'Acquisition refused; see RUNBOOK.md for quiesced recovery.');
97
121
  const HOSTNAME = os.hostname();
98
122
  /** Linux: field 22 of /proc/<pid>/stat (starttime). null elsewhere / on error. */
99
123
  const readProcStartTime = (pid) => {
@@ -114,16 +138,6 @@ const readProcStartTime = (pid) => {
114
138
  return null;
115
139
  }
116
140
  };
117
- /** true if the pid exists (signal 0). EPERM means it exists but isn't ours. */
118
- const pidAlive = (pid) => {
119
- try {
120
- process.kill(pid, 0);
121
- return true;
122
- }
123
- catch (err) {
124
- return err.code === 'EPERM';
125
- }
126
- };
127
141
  const buildRecord = () => ({
128
142
  v: LOCK_RECORD_VERSION,
129
143
  pid: process.pid,
@@ -134,8 +148,16 @@ const buildRecord = () => ({
134
148
  acquiredAt: new Date().toISOString(),
135
149
  });
136
150
  const readRecord = (lockPath) => {
151
+ let raw;
152
+ try {
153
+ raw = readFileSync(lockPath, 'utf8');
154
+ }
155
+ catch (err) {
156
+ if (err.code === 'ENOENT')
157
+ return null;
158
+ throw err; // An IO/permission failure is not evidence of a malformed orphan.
159
+ }
137
160
  try {
138
- const raw = readFileSync(lockPath, 'utf8');
139
161
  const parsed = JSON.parse(raw);
140
162
  // `typeof NaN === 'number'`, so a bare number check lets NaN/0/-1/Infinity/
141
163
  // fractional pids reach process.kill (#2658 review L4): a garbled or crafted
@@ -145,11 +167,12 @@ const readRecord = (lockPath) => {
145
167
  return null;
146
168
  if (typeof parsed.token !== 'string')
147
169
  return null;
170
+ // Preserve the existing ownership format: unknown versions or missing
171
+ // ancillary metadata must not turn a live/foreign holder into an orphan.
148
172
  return parsed;
149
173
  }
150
174
  catch {
151
- // Missing (won the race, file gone) or malformed/half-written → treat as
152
- // "no readable holder"; the caller retries the O_EXCL create.
175
+ // Malformed/half-written; only reclaim under the guard after the grace.
153
176
  return null;
154
177
  }
155
178
  };
@@ -165,73 +188,13 @@ const readRecord = (lockPath) => {
165
188
  const isStale = (holder) => {
166
189
  if (holder.hostname !== HOSTNAME)
167
190
  return false;
168
- if (!pidAlive(holder.pid))
191
+ if (!isProcessAlive(holder.pid))
169
192
  return true;
170
193
  const now = readProcStartTime(holder.pid);
171
194
  if (holder.startTime && now && holder.startTime !== now)
172
195
  return true; // pid reused
173
196
  return false;
174
197
  };
175
- /**
176
- * Reclaim a lock file we judged reclaimable — a dead holder (`expected` = its
177
- * record) or a malformed/unreadable crash-orphan (`expected` = null) — moving
178
- * the exact inode aside in ONE `rename` syscall to a token-unique name so two
179
- * waiters reclaiming the same orphan can't both win (the loser's rename ENOENTs).
180
- *
181
- * CRITICAL (#2658 review): the reclaim must not act on a STALE judgment. The
182
- * staleness decision (`isStale` / malformed-grace) happened a few syscalls ago;
183
- * a live writer may have O_EXCL-created its own lock at `lockPath` since. Blindly
184
- * renaming that live lock aside would delete it and admit a SECOND writer — the
185
- * exact double-writer this lock exists to prevent (reproduced: ~18%/round under
186
- * 4-way reclaim contention on the file backend). So:
187
- * 1. re-read `lockPath` immediately BEFORE the rename and confirm it still holds
188
- * exactly what we judged (same token, or still-unreadable) — shrinking the
189
- * window to the single gap between this read and the rename;
190
- * 2. after the rename, confirm what we ACTUALLY moved matches the judgment; if a
191
- * live lock slipped into that residual gap, RESTORE it (rename back) so its
192
- * holder is never displaced, and lose the reclaim.
193
- * A concurrent creator whose fresh lock the restore overwrites is caught by the
194
- * acquire loop's post-write read-back verify (see acquireViaFile), so it backs
195
- * off rather than proceeding as a second writer.
196
- *
197
- * Returns true if we won the reclaim (caller retries the create), false if we
198
- * lost the race or the judgment went stale (caller re-loops and re-reads).
199
- */
200
- const matchesJudgment = (record, expected) => expected === null ? record === null : record?.token === expected.token;
201
- const stealLock = (lockPath, me, expected) => {
202
- // (1) Re-verify the judgment still holds right before we move anything.
203
- if (!matchesJudgment(readRecord(lockPath), expected))
204
- return false;
205
- if (expected === null && !existsSync(lockPath))
206
- return false; // malformed → but now vanished
207
- const aside = `${lockPath}.dead.${me.token}`;
208
- try {
209
- renameSync(lockPath, aside);
210
- }
211
- catch (err) {
212
- if (err.code === 'ENOENT')
213
- return false; // another stealer won
214
- throw err;
215
- }
216
- // (2) Confirm what we moved is what we judged; if a live lock slipped into the
217
- // read→rename gap, put it back — a live holder must never be displaced.
218
- if (!matchesJudgment(readRecord(aside), expected)) {
219
- try {
220
- renameSync(aside, lockPath); // restore; an overwritten concurrent creator's read-back backs it off
221
- }
222
- catch {
223
- /* slot re-taken between our move and restore — leave it; we lost the reclaim */
224
- }
225
- return false;
226
- }
227
- try {
228
- unlinkSync(aside); // uniquely ours by token → safe; best-effort
229
- }
230
- catch {
231
- /* leftover .dead.<token> is inert (not analyze.lock, not swept) — harmless */
232
- }
233
- return true;
234
- };
235
198
  /**
236
199
  * Placeholder holder for an {@link IndexLockTimeoutError} thrown while the lock
237
200
  * file exists but no valid record can be read (malformed/partial), or it keeps
@@ -248,13 +211,68 @@ const unknownHolder = () => ({
248
211
  invocationId: '<unreadable>',
249
212
  acquiredAt: '',
250
213
  });
214
+ /** Remaining wait before the next guard-create poll, or throws the matching timeout. */
215
+ const remainingGuardCreateWaitMs = (args) => {
216
+ const workloadDeadline = args.startedAt + args.timeoutMs;
217
+ const guardDeadline = args.guardWaitSince + GUARD_TIMEOUT_MS;
218
+ const deadline = Math.min(workloadDeadline, guardDeadline, args.permissionDeadline);
219
+ if (args.now < deadline)
220
+ return deadline - args.now;
221
+ if (args.permissionError && args.now >= args.permissionDeadline)
222
+ throw args.permissionError;
223
+ if (args.code === 'EPERM')
224
+ throw args.err;
225
+ if (args.now >= guardDeadline) {
226
+ throw new IndexLockTimeoutError(unknownHolder(), args.now - args.startedAt, false, args.guardPath);
227
+ }
228
+ const remembered = args.lastLiveHolder ?? unknownHolder();
229
+ throw new IndexLockTimeoutError(remembered, args.now - args.startedAt, args.lastLiveHolder !== null);
230
+ };
231
+ /** Drop this attempt's guard. A throw here discards the pending handle. */
232
+ const releaseAcquisitionGuard = (guardPath, me, createdMain, lockPath) => {
233
+ try {
234
+ const guardRecord = readRecord(guardPath);
235
+ if (guardRecord && guardRecord.token !== me.token) {
236
+ throw unverifiedGuardError(guardPath);
237
+ }
238
+ if (guardRecord?.token === me.token) {
239
+ // Must complete before returning a workload handle or polling.
240
+ unlinkSync(guardPath);
241
+ return;
242
+ }
243
+ try {
244
+ lstatSync(guardPath);
245
+ }
246
+ catch (err) {
247
+ if (err.code === 'ENOENT') {
248
+ throw unverifiedGuardError(guardPath);
249
+ }
250
+ throw err;
251
+ }
252
+ // Exists but unreadable: this process created the name via O_EXCL.
253
+ // Drop it so a failed metadata write cannot leave a permanent orphan,
254
+ // then refuse this attempt.
255
+ unlinkSync(guardPath);
256
+ throw unverifiedGuardError(guardPath);
257
+ }
258
+ catch (guardError) {
259
+ // Roll back only the token-exact record this attempt created.
260
+ if (createdMain) {
261
+ try {
262
+ if (readRecord(lockPath)?.token === me.token)
263
+ unlinkSync(lockPath);
264
+ }
265
+ catch (cleanupError) {
266
+ throw new AggregateError([guardError, cleanupError], `Guard and workload-lock cleanup failed: ${guardPath}. Acquisition refused; see RUNBOOK.md for quiesced recovery.`);
267
+ }
268
+ }
269
+ throw guardError;
270
+ }
271
+ };
251
272
  /**
252
- * Filesystem-create error codes we tolerate by proceeding lock-free: a
253
- * read-only mount (EROFS) or a denied create (EACCES/EPERM). Such a filesystem
254
- * rejects every index WRITE in the same directory too, so no concurrent writer
255
- * can exist and the lock is moot — an already-indexed repo on a `:ro` mount
256
- * must still reach its `alreadyUpToDate` fast path (#2658). A genuinely-needed
257
- * write fails later exactly as it would have without the lock.
273
+ * Filesystem-create errors eligible for a read-only, non-owning handle when
274
+ * neither lock nor guard exists. Denied creation does NOT prove other writers
275
+ * lack access (ACLs may differ). Callers that write must reject lockFree handles.
258
276
  */
259
277
  export const LOCK_UNWRITABLE_CODES = new Set(['EROFS', 'EACCES', 'EPERM']);
260
278
  export const isLockUnwritableCode = (code) => code !== undefined && LOCK_UNWRITABLE_CODES.has(code);
@@ -267,6 +285,22 @@ const noopHandle = (record) => ({
267
285
  lockFree: true,
268
286
  release: () => { },
269
287
  });
288
+ const deniedCreateHandle = (lockPath, record, error) => {
289
+ // Do not turn an existing (even malformed/unreadable) owner into permission
290
+ // to proceed. lstat also sees dangling links; only ENOENT proves absence.
291
+ for (const candidate of [lockPath, lockGuardPath(lockPath)]) {
292
+ try {
293
+ lstatSync(candidate);
294
+ }
295
+ catch (err) {
296
+ if (err.code === 'ENOENT')
297
+ continue;
298
+ throw err;
299
+ }
300
+ throw error;
301
+ }
302
+ return noopHandle(record);
303
+ };
270
304
  /**
271
305
  * Delete orphaned build/staging artifacts left in the lock directory by a
272
306
  * crashed prior writer. Safe precisely because we hold the exclusive lock: no
@@ -326,36 +360,31 @@ const resolveTimeoutMs = (opt) => {
326
360
  const n = Number(env);
327
361
  return Number.isFinite(n) ? n : DEFAULT_TIMEOUT_MS;
328
362
  })();
363
+ // Explicit NaN is a number, so it used to skip the env finite-check and
364
+ // poison every deadline (`startedAt + NaN`). Match the env fallback.
365
+ if (Number.isNaN(raw))
366
+ return DEFAULT_TIMEOUT_MS;
329
367
  return raw <= 0 ? Number.POSITIVE_INFINITY : raw;
330
368
  };
331
- /**
332
- * Acquire the exclusive write lock for `lockDir` (the resolved index slot
333
- * directory, e.g. `<repo>/.gitnexus` or `<repo>/.gitnexus/branches/<slug>`).
334
- *
335
- * Blocks until the lock is held (waiting only on live holders, stealing dead
336
- * ones immediately), then sweeps orphaned staging files under the lock and
337
- * returns a handle. Rejects with `IndexLockTimeoutError` if `timeoutMs` is
338
- * exceeded while a live holder still holds the lock.
339
- */
340
369
  /**
341
370
  * File-based (O_EXCL pidfile) backend. The portable fallback used on platforms
342
371
  * without the socket backend (macOS/BSD) or when the OS socket lock is
343
- * unavailable. Carries the pid-liveness staleness, atomic rename-steal reclaim,
344
- * bounded malformed-file handling, and read-only tolerance. Its stale-takeover
345
- * has an irreducible (narrow) race — see the module header — which is why the
346
- * socket backend is preferred where available.
372
+ * unavailable. All inspection/reclaim/create/verify operations are serialized
373
+ * by a sibling O_EXCL guard. No waiter ever removes a guard, regardless of age,
374
+ * pid liveness, or malformed metadata. See RUNBOOK.md for orphan recovery.
347
375
  */
348
376
  const acquireViaFile = async (lockDir, me, opts) => {
377
+ const lockPath = path.join(lockDir, LOCK_FILENAME);
349
378
  try {
350
379
  mkdirSync(lockDir, { recursive: true });
351
380
  }
352
381
  catch (err) {
353
- // Read-only / denied filesystem → proceed lock-free (see LOCK_UNWRITABLE_CODES).
382
+ // Read-only / denied mkdir → lockFree only when neither lock nor guard exists.
354
383
  if (isLockUnwritableCode(err.code))
355
- return noopHandle(me);
384
+ return deniedCreateHandle(lockPath, me, err);
356
385
  throw err;
357
386
  }
358
- const lockPath = path.join(lockDir, LOCK_FILENAME);
387
+ const guardPath = lockGuardPath(lockPath);
359
388
  const pollMs = opts.pollMs ?? DEFAULT_POLL_MS;
360
389
  const timeoutMs = resolveTimeoutMs(opts.timeoutMs);
361
390
  const startedAt = Date.now();
@@ -364,63 +393,176 @@ const acquireViaFile = async (lockDir, me, opts) => {
364
393
  // When the lock file exists but has no readable record, the timestamp we
365
394
  // first observed it unreadable — used to reclaim a crash-orphan after a grace.
366
395
  let malformedSince = null;
396
+ let guardWaitSince = null;
397
+ let permissionWaitSince = null;
398
+ let permissionError;
399
+ // Last live workload holder observed while we held the inspect guard. Used
400
+ // when the overall wait budget expires during a brief peer inspect (EEXIST)
401
+ // so we do not mislabel ordinary contention as an orphan guard.
402
+ let lastLiveHolder = null;
367
403
  for (;;) {
404
+ const permissionDeadline = permissionWaitSince === null
405
+ ? Number.POSITIVE_INFINITY
406
+ : permissionWaitSince + GUARD_TIMEOUT_MS;
407
+ if (Date.now() >= Math.min(startedAt + timeoutMs, permissionDeadline) && permissionError) {
408
+ throw permissionError;
409
+ }
410
+ let guardFd;
411
+ try {
412
+ guardFd = openSync(guardPath, 'wx');
413
+ }
414
+ catch (err) {
415
+ const code = err.code;
416
+ // Windows can report EPERM while an unlinked guard is delete-pending.
417
+ // Retry under the same bounded budget, never interpret it as ownership.
418
+ if (code === 'EEXIST' || code === 'EPERM') {
419
+ const now = Date.now();
420
+ guardWaitSince ??= now;
421
+ const remainingMs = remainingGuardCreateWaitMs({
422
+ now,
423
+ startedAt,
424
+ timeoutMs,
425
+ guardWaitSince,
426
+ permissionDeadline,
427
+ permissionError,
428
+ code,
429
+ err,
430
+ guardPath,
431
+ lastLiveHolder,
432
+ });
433
+ await sleep(jitteredDelay(pollMs, remainingMs, 0));
434
+ continue;
435
+ }
436
+ if (isLockUnwritableCode(code))
437
+ return deniedCreateHandle(lockPath, me, err);
438
+ throw err;
439
+ }
440
+ guardWaitSince = null;
441
+ // Metadata is diagnostic. Cleanup unlinks our token-exact record, or a
442
+ // self-created unreadable leftover (then refuses this attempt). A foreign
443
+ // token is never removed.
444
+ let holder = null;
445
+ let createdMain = false;
368
446
  try {
369
- // O_WRONLY | O_CREAT | O_EXCL — the atomic arbiter of ownership.
370
- const fd = openSync(lockPath, 'wx');
371
447
  try {
372
- writeSync(fd, JSON.stringify(me));
448
+ writeSync(guardFd, JSON.stringify(me));
373
449
  }
374
450
  finally {
375
- closeSync(fd);
451
+ closeSync(guardFd);
376
452
  }
377
- // Read-back verify (#2658 review L5): if this process stalled (a >graceMs
378
- // GC pause) between the O_EXCL create of the *empty* file and the write
379
- // above, a waiter could have reclaimed the empty file (renamed it aside)
380
- // and O_EXCL-created its own lock at `lockPath`. Our write then landed on
381
- // the renamed-aside inode, not `lockPath`. Confirm `lockPath` still carries
382
- // our token before claiming ownership; if it was stolen, contend normally.
383
- const confirmed = readRecord(lockPath);
384
- if (!confirmed || confirmed.token !== me.token)
385
- continue;
386
- return {
387
- record: me,
388
- release: () => {
389
- const current = readRecord(lockPath);
390
- if (current && current.token !== me.token)
391
- return; // no longer ours
453
+ holder = readRecord(lockPath);
454
+ let tryCreate = false;
455
+ if (holder) {
456
+ malformedSince = null;
457
+ if (isStale(holder)) {
458
+ opts.log?.(`Reclaiming stale index lock from dead analyze (pid ${holder.pid}, ` +
459
+ `invocation ${holder.invocationId}).`);
460
+ unlinkSync(lockPath);
461
+ tryCreate = true;
462
+ holder = null;
463
+ }
464
+ }
465
+ else {
466
+ tryCreate = true;
467
+ }
468
+ if (tryCreate) {
469
+ let fd;
470
+ try {
471
+ fd = openSync(lockPath, 'wx');
472
+ createdMain = true;
473
+ }
474
+ catch (err) {
475
+ const code = err.code;
476
+ if (code === 'EEXIST') {
477
+ // Exclusive create is the presence check — do not existsSync first.
478
+ const current = readRecord(lockPath);
479
+ if (current === null) {
480
+ malformedSince ??= Date.now();
481
+ if (Date.now() - malformedSince >= malformedGraceMs(pollMs)) {
482
+ opts.log?.('Reclaiming a malformed/partial index lock file (no readable owner record).');
483
+ unlinkSync(lockPath);
484
+ malformedSince = null;
485
+ }
486
+ }
487
+ else {
488
+ holder = current;
489
+ }
490
+ }
491
+ else if (code === 'EPERM') {
492
+ throw err;
493
+ }
494
+ else if (isLockUnwritableCode(code)) {
495
+ return deniedCreateHandle(lockPath, me, err);
496
+ }
497
+ else {
498
+ throw err;
499
+ }
500
+ }
501
+ if (createdMain && fd !== undefined) {
392
502
  try {
393
- unlinkSync(lockPath);
503
+ writeSync(fd, JSON.stringify(me));
394
504
  }
395
- catch {
396
- /* already gone */
505
+ finally {
506
+ closeSync(fd);
397
507
  }
398
- },
399
- };
400
- }
401
- catch (err) {
402
- const code = err.code;
403
- if (code === 'EEXIST') {
404
- // fall through to holder inspection / wait / reclaim below
508
+ if (readRecord(lockPath)?.token !== me.token) {
509
+ throw new Error(`Index lock verification failed: ${lockPath}`);
510
+ }
511
+ let released = false;
512
+ return {
513
+ record: me,
514
+ release: () => {
515
+ if (released)
516
+ return;
517
+ released = true;
518
+ // No async boundary: a cooperating contender cannot replace a live
519
+ // owner's record before this one-shot unlink. Unknown is not ours.
520
+ try {
521
+ if (readRecord(lockPath)?.token !== me.token)
522
+ return;
523
+ unlinkSync(lockPath);
524
+ }
525
+ catch {
526
+ /* best-effort on exit; never retry against a successor */
527
+ }
528
+ },
529
+ };
530
+ }
405
531
  }
406
- else if (isLockUnwritableCode(code)) {
407
- return noopHandle(me); // read-only / denied → proceed lock-free
532
+ permissionWaitSince = null;
533
+ permissionError = undefined;
534
+ }
535
+ catch (error) {
536
+ if (!createdMain && error.code === 'EPERM') {
537
+ // A releasing owner may leave the main file delete-pending on Windows.
538
+ // Release our guard in finally and retry; never reclaim an unreadable file.
539
+ permissionWaitSince ??= Date.now();
540
+ permissionError = error;
541
+ holder = null;
542
+ if (Date.now() >= Math.min(startedAt + timeoutMs, permissionWaitSince + GUARD_TIMEOUT_MS)) {
543
+ throw error;
544
+ }
408
545
  }
409
546
  else {
410
- throw err;
547
+ if (createdMain) {
548
+ try {
549
+ if (readRecord(lockPath)?.token === me.token)
550
+ unlinkSync(lockPath);
551
+ }
552
+ catch (cleanupError) {
553
+ throw new AggregateError([error, cleanupError], `Workload-lock cleanup failed: ${lockPath}. Acquisition refused; see RUNBOOK.md for quiesced recovery.`);
554
+ }
555
+ }
556
+ throw error;
411
557
  }
412
558
  }
413
- const holder = readRecord(lockPath);
559
+ finally {
560
+ releaseAcquisitionGuard(guardPath, me, createdMain, lockPath);
561
+ }
414
562
  const waited = Date.now() - startedAt;
415
563
  if (holder) {
416
- malformedSince = null;
417
- if (isStale(holder)) {
418
- opts.log?.(`Reclaiming stale index lock from dead analyze (pid ${holder.pid}, ` +
419
- `invocation ${holder.invocationId}).`);
420
- stealLock(lockPath, me, holder); // reclaim ONLY this dead record; live locks are never stolen
421
- continue;
422
- }
423
564
  // Live holder → wait.
565
+ lastLiveHolder = holder;
424
566
  if (!announcedWait) {
425
567
  announcedWait = true;
426
568
  opts.onWaitStart?.(holder);
@@ -438,30 +580,12 @@ const acquireViaFile = async (lockDir, me, opts) => {
438
580
  await sleep(jitteredDelay(pollMs, timeoutMs, waited));
439
581
  continue;
440
582
  }
441
- // holder === null: the lock file is either gone (vanished between the failed
442
- // create and our read) or present-but-unreadable (a crash between the
443
- // O_EXCL create and the record write, or a partial write). NEVER hot-loop
444
- // here — both branches are bounded by sleep + timeout.
445
- if (!existsSync(lockPath)) {
446
- malformedSince = null; // genuinely vanished → the next create likely wins
447
- if (waited >= timeoutMs)
448
- throw new IndexLockTimeoutError(unknownHolder(), waited, false);
449
- await sleep(jitteredDelay(pollMs, timeoutMs, waited));
450
- continue;
451
- }
452
- // Malformed orphan present. Reclaim only after a grace, so a live owner's
453
- // microsecond create→write window is never mistaken for a crash.
454
- if (malformedSince === null)
455
- malformedSince = Date.now();
456
- if (Date.now() - malformedSince >= malformedGraceMs(pollMs)) {
457
- opts.log?.('Reclaiming a malformed/partial index lock file (no readable owner record).');
458
- stealLock(lockPath, me, null); // reclaim ONLY while still unreadable; a live lock written since is left
459
- malformedSince = null;
460
- continue;
461
- }
462
583
  if (waited >= timeoutMs)
463
584
  throw new IndexLockTimeoutError(unknownHolder(), waited, false);
464
- await sleep(jitteredDelay(pollMs, timeoutMs, waited));
585
+ const waitCeiling = Math.min(timeoutMs, permissionWaitSince === null
586
+ ? Number.POSITIVE_INFINITY
587
+ : permissionWaitSince + GUARD_TIMEOUT_MS - startedAt);
588
+ await sleep(jitteredDelay(pollMs, waitCeiling, waited));
465
589
  }
466
590
  };
467
591
  /** Signals that the OS socket backend can't be used here (e.g. abstract
@@ -648,10 +772,10 @@ export const acquireIndexLock = async (lockDir, opts = {}) => {
648
772
  else {
649
773
  handle = await acquireViaFile(lockDir, me, opts);
650
774
  }
651
- // Reclaim crashed-build staging orphans while we hold the lock. Best-effort:
652
- // a read-only mount (no orphans reachable) just no-ops.
775
+ // Without ownership, a staging file may belong to an active writer.
653
776
  try {
654
- sweepStagingArtifacts(lockDir, opts.log);
777
+ if (!handle.lockFree)
778
+ sweepStagingArtifacts(lockDir, opts.log);
655
779
  }
656
780
  catch {
657
781
  /* best-effort */
@@ -21,7 +21,7 @@ import { stripWindowsLongPathPrefix } from '../lib/utils.js';
21
21
  import { writeFileAtomic } from './fs-atomic.js';
22
22
  import { getGlobalDir } from './global-dir.js';
23
23
  import { logger } from '../core/logger.js';
24
- import { acquireIndexLock, IndexLockTimeoutError } from './index-lock.js';
24
+ import { acquireIndexLock, IndexLockTimeoutError, requireExclusiveIndexLock, } from './index-lock.js';
25
25
  import { branchSlug, BRANCHES_DIR, resolveBranchPlacement, } from './branch-index.js';
26
26
  import { GITNEXUS_DIR, INDEX_METADATA_FILE, LEGACY_METADATA_FILE, getStoragePath, isMissingFilesystemError, loadMeta, tryReadMetaFile, } from './repo-meta.js';
27
27
  // Re-export the #2106 branch primitives (extracted to branch-index.ts, R10) so
@@ -471,7 +471,8 @@ const REGISTRY_LOCK_TIMEOUT_MS = 5_000;
471
471
  * The registry is shared by every indexed repository, so per-index locks do
472
472
  * not protect this file. Reuse the cross-platform index lock primitive with a
473
473
  * registry-private lock namespace; the handle is kernel-owned on supported
474
- * platforms and crash-reclaimable by the existing fallback.
474
+ * platforms. The file fallback reclaims dead workload holders; an orphan
475
+ * acquisition guard requires quiesced recovery (RUNBOOK.md).
475
476
  *
476
477
  * On timeout the transaction fails closed: continuing unlocked would reintroduce
477
478
  * the lost-update race this lock exists to prevent and can silently discard a
@@ -495,6 +496,7 @@ const withRegistryLock = async (operation) => {
495
496
  throw err;
496
497
  }
497
498
  try {
499
+ requireExclusiveIndexLock(lock, 'Cannot acquire the global registry lock; refusing an unlocked registry transaction.');
498
500
  return await operation();
499
501
  }
500
502
  finally {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "gitnexus",
3
- "version": "1.6.12-rc.38",
3
+ "version": "1.6.12-rc.39",
4
4
  "description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
5
5
  "author": "Abhigyan Patwari",
6
6
  "license": "PolyForm-Noncommercial-1.0.0",