gitnexus 1.6.13-rc.39 → 1.6.13-rc.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -212,25 +212,25 @@ Note that the bundled Graphology path is no longer the slow option it once was:
212
212
 
213
213
  Your AI agent gets **17 tools** (15 per-repo + 2 group) automatically:
214
214
 
215
- | Tool | What It Does |
216
- | ---------------- | ---------------------------------------------------------------------- |
217
- | `list_repos` | Discover all indexed repositories (paginated — `limit`/`offset`) |
218
- | `query` | Process-grouped hybrid search (BM25 + semantic + RRF); optional `chain_depth` expands each result's call chain |
215
+ | Tool | What It Does |
216
+ | ---------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- |
217
+ | `list_repos` | Discover all indexed repositories (paginated — `limit`/`offset`) |
218
+ | `query` | Process-grouped hybrid search (BM25 + semantic + RRF); optional `chain_depth` expands each result's call chain |
219
219
  | `context` | 360-degree symbol view — categorized refs, process participation, HTTP routes, `is_entry_point` flag; optional `chain_depth` call-chain expansion |
220
- | `impact` | Blast radius analysis with depth grouping and confidence |
221
- | `trace` | Shortest directed path between two symbols (call + class-member edges) |
222
- | `detect_changes` | Git-diff impact — maps changed lines to affected processes |
223
- | `check` | Read-only structural checks against the indexed graph |
224
- | `rename` | Multi-file coordinated rename with graph + text search |
225
- | `cypher` | Raw Cypher graph queries |
226
- | `route_map` | API route map — which components fetch which endpoints, and handlers |
227
- | `tool_map` | MCP/RPC tool definitions — where they're defined and handled |
228
- | `shape_check` | Validate API response shapes against consumers' property accesses |
229
- | `api_impact` | Pre-change impact report for an API route handler |
230
- | `explain` | Explain persisted taint findings (source→sink flows, `--pdg` indexes) |
231
- | `pdg_query` | Query control/data dependence at statement level (`--pdg` indexes) |
232
- | `group_list` | List configured repository groups |
233
- | `group_sync` | Rebuild a group's Contract Registry and cross-repo links |
220
+ | `impact` | Blast radius analysis with depth grouping and confidence |
221
+ | `trace` | Shortest directed path between two symbols (call + class-member edges) |
222
+ | `detect_changes` | Git-diff impact — maps changed lines to affected processes |
223
+ | `check` | Read-only structural checks against the indexed graph |
224
+ | `rename` | Multi-file coordinated rename with graph + text search |
225
+ | `cypher` | Raw Cypher graph queries |
226
+ | `route_map` | API route map — which components fetch which endpoints, and handlers |
227
+ | `tool_map` | MCP/RPC tool definitions — where they're defined and handled |
228
+ | `shape_check` | Validate API response shapes against consumers' property accesses |
229
+ | `api_impact` | Pre-change impact report for an API route handler |
230
+ | `explain` | Explain persisted taint findings (source→sink flows, `--pdg` indexes) |
231
+ | `pdg_query` | Query control/data dependence at statement level (`--pdg` indexes) |
232
+ | `group_list` | List configured repository groups |
233
+ | `group_sync` | Rebuild a group's Contract Registry and cross-repo links |
234
234
 
235
235
  > Read-only tools can omit `repo` when one repo is indexed, an MCP default is configured, or the GitNexus process cwd is inside a registered path without crossing into an unindexed nested Git checkout. Otherwise—and for mutating tools with multiple indexed repos and no MCP default—specify it explicitly: `query({search_query: "auth", repo: "my-app"})`. Per-repo tools also take an optional `branch` for indexes pinned with `gitnexus analyze --branch`; omitting it queries the workspace index, which follows your checked-out working tree. `explain` and `pdg_query` need an index built with `gitnexus analyze --pdg`.
236
236
 
@@ -279,6 +279,7 @@ gitnexus analyze --spring-actuator ./actuator # Enrich with local Spring Boot A
279
279
  gitnexus analyze --verbose # Log skipped files when parsers are unavailable
280
280
  gitnexus analyze --max-file-size 1024 # Skip files larger than N KB (default: 512, cap: 32768)
281
281
  gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses
282
+ gitnexus analyze --memory-budget 3000 # Main-thread V8 heap in MB (>= 200); overrides the auto-sizer and any --max-old-space-size pin
282
283
  gitnexus analyze --wal-checkpoint-threshold 67108864 # 64 MiB. Control LadybugDB WAL auto-checkpoint threshold (default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB)
283
284
  gitnexus auto-sync [init|start|restart|stop|status|reset] # Scheduled remote clone/pull + analyze from GITNEXUS_HOME/watch_config.yml
284
285
  gitnexus mcp # Start MCP server (stdio) — serves all indexed repos
@@ -754,6 +755,11 @@ If analyze says the repository doesn't fit, do what the message says:
754
755
  - **The machine is the ceiling**: shrink the scope (exclude generated or
755
756
  vendored directories, below) or use a machine with more RAM.
756
757
 
758
+ To set the main-thread heap yourself on a memory-constrained host, pass
759
+ `--memory-budget <mb>`: analyze re-runs with exactly that V8 heap, overriding
760
+ the auto-sizer and any `--max-old-space-size` pin. It sizes the main thread
761
+ only; parse workers keep their own caps.
762
+
757
763
  Escape hatches (`GITNEXUS_MEMORY=off` to decline the autopilot,
758
764
  `GITNEXUS_WORKER_HEAP_MB` to size workers yourself) are listed in the
759
765
  environment-variable table below —
@@ -839,6 +845,7 @@ Four env vars expose the pool's resilience layers (respawn budget, cumulative-ti
839
845
  | `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. |
840
846
  | `GITNEXUS_MEMORY` | `off` | unset (autopilot on) | `off` declines GitNexus's memory autopilot: analyze will neither re-run itself with a RAM-aware heap cap nor abort the parse before V8 enters its ineffective-mark-compact death spiral. Use it when you want to drive memory manually; to simply pin a heap size, pass Node's own `--max-old-space-size`, which is already honoured as your decision. |
841
847
  | `GITNEXUS_WORKER_HEAP_MB` | `clamp(512, RAM/2/poolSize, 4096)` | Per-worker V8 old-generation heap cap (#2649). Bounds pool RSS on large repos; a worker exceeding it dies with a real heap error handled by quarantine/respawn. |
848
+ | `GITNEXUS_HEAP_LIMIT_SOURCE` | unset | Set by analyze itself (`budget` or `auto`) to record where the heap limit came from, so out-of-memory advice points at `--memory-budget` rather than a `--max-old-space-size` pin. | Never — internal; set `--memory-budget` instead. |
842
849
  | `GITNEXUS_SERVER_ANALYZE_HEAP_MB` | `min(8192, auto cap)` | Heap for the web/MCP server's forked analyze worker (#2649). Defaults to the historical 8192 MB bounded by the machine/container's RAM-aware auto cap; set an absolute MB value to override. |
843
850
  | `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. |
844
851
 
@@ -117,6 +117,14 @@ export interface AnalyzeOptions {
117
117
  workerTimeout?: string;
118
118
  /** Control LadybugDB WAL auto-checkpoint threshold during analyze. */
119
119
  walCheckpointThreshold?: string;
120
+ /**
121
+ * `--memory-budget <mb>` (#3137): the main-thread V8 heap limit in MB.
122
+ * `ensureHeap` applies it through the existing heap respawn, replacing the
123
+ * RAM-aware auto cap and any `--max-old-space-size` pin, so the #2649
124
+ * guards read it as the live limit. Parse workers keep their own heap caps.
125
+ * Integer, minimum 200; CLI-only (not a `.gitnexusrc` key).
126
+ */
127
+ memoryBudget?: string;
120
128
  /** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */
121
129
  workers?: string;
122
130
  /** Process-detection process cap. Positive integer string; `0` is invalid. */
@@ -11,6 +11,7 @@ import { GITNEXUS_DIR } from '../storage/repo-meta.js';
11
11
  import { loadAnalyzeConfigStrict, mergeAnalyzeOptions, validateBranchName, } from './analyze-config.js';
12
12
  import { ensureHeap } from './analyze.js';
13
13
  import { cliError, cliInfo, cliWarn } from './cli-message.js';
14
+ import { parseIntegerOption } from './int-option.js';
14
15
  import { formatInvalidProcessDetectionOverride, parseProcessDetectionBudgetStrings, } from '../core/ingestion/process-detection-budget.js';
15
16
  import { WATCH_FULL_REFRESH_PATH, WatchRefreshQueue, } from './watch-queue.js';
16
17
  const DEFAULT_DEBOUNCE_MS = 300;
@@ -54,9 +55,7 @@ function setEnvironment(name, value) {
54
55
  function positiveInteger(value, flag, maximum) {
55
56
  if (value === undefined)
56
57
  return undefined;
57
- const parsed = Number(value);
58
- if (!Number.isInteger(parsed) || parsed < 1)
59
- throw new Error(`${flag} must be a positive integer`);
58
+ const parsed = parseIntegerOption(value, flag, { minimum: 1 });
60
59
  if (maximum !== undefined && parsed > maximum) {
61
60
  throw new Error(`${flag} must not exceed ${maximum}`);
62
61
  }
@@ -329,8 +328,9 @@ export async function startWatchFileLoop(repoPath, debounceMs, refresh, onError,
329
328
  };
330
329
  }
331
330
  export async function watchCommandWithRunnerIdentity(runnerIdentityAtBootstrap, inputPath, cliOptions = {}) {
332
- if (await ensureHeap({ cleanForwardedTermination: true }))
331
+ if (await ensureHeap({ cleanForwardedTermination: true, memoryBudget: cliOptions.memoryBudget })) {
333
332
  return;
333
+ }
334
334
  const requestedRepoPath = inputPath ? path.resolve(inputPath) : getGitRoot(process.cwd());
335
335
  if (requestedRepoPath === null || !hasGitDir(requestedRepoPath)) {
336
336
  cliError(' gitnexus analyze --watch requires a Git repository.');
@@ -25,21 +25,69 @@ export declare function computeHeapCapMb(totalBytes: number, constrainedBytes: n
25
25
  * later-flag-wins semantics when NODE_OPTIONS repeats a flag.
26
26
  */
27
27
  export declare function parseMaxOldSpaceMb(nodeOptions: string): number | null;
28
+ export declare function forwardedSignalExitCode(signal: NodeJS.Signals, cleanTermination: boolean): number;
29
+ /** Old-space and semi-space flags (MB) that give a child a V8 heap of exactly the budget. */
30
+ export interface BudgetHeapSizing {
31
+ oldSpaceMb: number;
32
+ semiSpaceMb: number;
33
+ }
34
+ /**
35
+ * Size the respawn flags so the child's `heap_size_limit` equals the budget
36
+ * (#3137). V8's limit is old space plus three semi-spaces, and V8 rounds the
37
+ * semi-space up to a power of two (measured on Node 22.18: semi 10 → 16), so
38
+ * the semi-space is budget/25 rounded up the same way, capped at the usual
39
+ * 128MB, and the old space takes the rest: 2000 → 1616 + 3 × 128, 200 → 176 + 3 × 8.
40
+ */
41
+ export declare function budgetHeapSizing(budgetMb: number): BudgetHeapSizing;
42
+ export interface BudgetHeapInput {
43
+ budgetMb: number;
44
+ execArgv: readonly string[];
45
+ nodeOptions: string;
46
+ /** The RAM-aware auto cap the budget replaces. */
47
+ autoCapMb: number;
48
+ /** `GITNEXUS_MEMORY=off`: names Node's default limit as the one replaced. */
49
+ autopilotDisabled: boolean;
50
+ /** Inherited `GITNEXUS_HEAP_LIMIT_SOURCE`; `budget` marks a budget-respawned child. */
51
+ inheritedSource: string | undefined;
52
+ }
53
+ export interface BudgetHeapDecision {
54
+ respawn: boolean;
55
+ sizing: BudgetHeapSizing;
56
+ /** At most one warning line for the operator; absent in a budget-respawned child. */
57
+ warning?: string;
58
+ }
59
+ /**
60
+ * The `--memory-budget` heap decision (#3137), kept pure so every branch is
61
+ * testable without a process. The effective requested old space is the last
62
+ * `--max-old-space-size` in execArgv, else in NODE_OPTIONS (every spelling
63
+ * `parseMaxOldSpaceMb` accepts). Only the budget-respawned child keeps
64
+ * running; anything else respawns once at the budget. The budget's own flags
65
+ * follow the user's in the child, so the child always resolves to "at the
66
+ * budget".
67
+ */
68
+ export declare function resolveBudgetHeap(input: BudgetHeapInput): BudgetHeapDecision;
28
69
  /** Re-exec the process with the RAM-aware auto heap cap + larger semi-space/stack
29
- * if we're currently below that.
70
+ * if we're currently below that, or at the `--memory-budget` heap (#3137).
30
71
  *
31
- * Heap-source precedence (#2649):
32
- * - an explicit per-invocation `--max-old-space-size` (execArgv) always wins;
72
+ * Heap-source precedence (#2649, #3137):
73
+ * - an explicit `--memory-budget` wins over everything, including
74
+ * `GITNEXUS_MEMORY=off` and any `--max-old-space-size` pin;
75
+ * - an explicit per-invocation `--max-old-space-size` (execArgv) wins next;
33
76
  * - `GITNEXUS_MEMORY=off` declines the memory autopilot entirely;
34
77
  * - an ambient NODE_OPTIONS heap >= the auto cap is honored as-is;
35
78
  * - an ambient NODE_OPTIONS heap BELOW the auto cap is treated as an
36
79
  * inherited environment default (devcontainers/CI export one for other
37
80
  * tooling), not a deliberate per-run choice: warn and respawn with the
38
81
  * auto cap. Pre-#2649 this returned early and large repos then OOM'd on
39
- * whatever heap the environment happened to specify. */
40
- export declare function forwardedSignalExitCode(signal: NodeJS.Signals, cleanTermination: boolean): number;
82
+ * whatever heap the environment happened to specify.
83
+ *
84
+ * Paths managed by the auto-sizer or `--memory-budget` record the source in
85
+ * `GITNEXUS_HEAP_LIMIT_SOURCE` (this process when kept, the child env when
86
+ * respawned) so the parse phase's remedy text advises by source. Explicit
87
+ * heap pins and `GITNEXUS_MEMORY=off` intentionally leave it unset. */
41
88
  export declare function ensureHeap(options?: {
42
89
  cleanForwardedTermination?: boolean;
90
+ memoryBudget?: string;
43
91
  }): Promise<boolean>;
44
92
  /**
45
93
  * CLI `analyze` flag shape. Defined in `./analyze-options.js` so
@@ -30,7 +30,8 @@ import { warnMissingOptionalGrammars, getOptionalGrammarExtensions } from './opt
30
30
  import { glob } from 'glob';
31
31
  import fs from 'fs/promises';
32
32
  import { cliError, cliWarn } from './cli-message.js';
33
- import { heapCapMbFor, memoryAutopilotDisabled } from '../core/ingestion/utils/effective-ram.js';
33
+ import { IntegerOptionError, parseIntegerOption, parseMemoryBudgetMb } from './int-option.js';
34
+ import { HEAP_LIMIT_SOURCE_ENV, heapCapMbFor, memoryAutopilotDisabled, } from '../core/ingestion/utils/effective-ram.js';
34
35
  import { EMBEDDING_DIMS_ERROR, normalizeEmbeddingDims } from './embedding-dims.js';
35
36
  import { formatElapsed } from './format-elapsed.js';
36
37
  import { isHfDownloadFailure } from '../core/embeddings/hf-env.js';
@@ -405,27 +406,22 @@ const RECOMMENDED_WAL_CHECKPOINT_THRESHOLD = 64 * 1024 * 1024;
405
406
  * later-flag-wins semantics when NODE_OPTIONS repeats a flag.
406
407
  */
407
408
  export function parseMaxOldSpaceMb(nodeOptions) {
408
- // V8 accepts `-` and `_` interchangeably in flag names, and Node accepts a
409
- // space-separated value in NODE_OPTIONS — honor every spelling of the pin
410
- // instead of silently overriding it (#2649 review).
411
- const matches = [...nodeOptions.matchAll(/--max[-_]old[-_]space[-_]size(?:=|\s+)(\d+)/g)];
409
+ return parseV8SizeFlagMb(nodeOptions, 'max-old-space-size');
410
+ }
411
+ /**
412
+ * Last value (MB) of a V8 `--<name>` size flag in a flag string. V8 accepts
413
+ * `-` and `_` interchangeably in flag names, and Node accepts a
414
+ * space-separated value — honor every spelling instead of silently
415
+ * overriding it (#2649 review).
416
+ */
417
+ function parseV8SizeFlagMb(flags, name) {
418
+ const stem = name.split('-').join('[-_]');
419
+ const matches = [...flags.matchAll(new RegExp(`--${stem}(?:=|\\s+)(\\d+)`, 'g'))];
412
420
  if (matches.length === 0)
413
421
  return null;
414
422
  const mb = Number(matches[matches.length - 1][1]);
415
423
  return Number.isFinite(mb) && mb > 0 ? mb : null;
416
424
  }
417
- /** Re-exec the process with the RAM-aware auto heap cap + larger semi-space/stack
418
- * if we're currently below that.
419
- *
420
- * Heap-source precedence (#2649):
421
- * - an explicit per-invocation `--max-old-space-size` (execArgv) always wins;
422
- * - `GITNEXUS_MEMORY=off` declines the memory autopilot entirely;
423
- * - an ambient NODE_OPTIONS heap >= the auto cap is honored as-is;
424
- * - an ambient NODE_OPTIONS heap BELOW the auto cap is treated as an
425
- * inherited environment default (devcontainers/CI export one for other
426
- * tooling), not a deliberate per-run choice: warn and respawn with the
427
- * auto cap. Pre-#2649 this returned early and large repos then OOM'd on
428
- * whatever heap the environment happened to specify. */
429
425
  export function forwardedSignalExitCode(signal, cleanTermination) {
430
426
  if (cleanTermination)
431
427
  return 0;
@@ -435,7 +431,118 @@ export function forwardedSignalExitCode(signal, cleanTermination) {
435
431
  return 143;
436
432
  return 1;
437
433
  }
434
+ /**
435
+ * Size the respawn flags so the child's `heap_size_limit` equals the budget
436
+ * (#3137). V8's limit is old space plus three semi-spaces, and V8 rounds the
437
+ * semi-space up to a power of two (measured on Node 22.18: semi 10 → 16), so
438
+ * the semi-space is budget/25 rounded up the same way, capped at the usual
439
+ * 128MB, and the old space takes the rest: 2000 → 1616 + 3 × 128, 200 → 176 + 3 × 8.
440
+ */
441
+ export function budgetHeapSizing(budgetMb) {
442
+ const scaled = Math.max(1, Math.floor(budgetMb / 25));
443
+ const semiSpaceMb = Math.min(SEMI_SPACE_MB, 2 ** Math.ceil(Math.log2(scaled)));
444
+ return { oldSpaceMb: budgetMb - 3 * semiSpaceMb, semiSpaceMb };
445
+ }
446
+ /**
447
+ * The `--memory-budget` heap decision (#3137), kept pure so every branch is
448
+ * testable without a process. The effective requested old space is the last
449
+ * `--max-old-space-size` in execArgv, else in NODE_OPTIONS (every spelling
450
+ * `parseMaxOldSpaceMb` accepts). Only the budget-respawned child keeps
451
+ * running; anything else respawns once at the budget. The budget's own flags
452
+ * follow the user's in the child, so the child always resolves to "at the
453
+ * budget".
454
+ */
455
+ export function resolveBudgetHeap(input) {
456
+ const { budgetMb, autoCapMb } = input;
457
+ const sizing = budgetHeapSizing(budgetMb);
458
+ const execFlags = input.execArgv.join(' ');
459
+ const pinnedMb = parseMaxOldSpaceMb(execFlags) ?? parseMaxOldSpaceMb(input.nodeOptions);
460
+ const semiSpaceMb = parseV8SizeFlagMb(execFlags, 'max-semi-space-size') ??
461
+ parseV8SizeFlagMb(input.nodeOptions, 'max-semi-space-size');
462
+ // Only a budget-respawned child carries both the budget's old space and its
463
+ // semi-space, so only it is exactly at the budget. A pin that merely equals
464
+ // the budget leaves V8's young generation on top (a 2000MB pin is a 2048MB
465
+ // heap), so that process respawns once like any other. The respawning
466
+ // parent already logged; the child stays quiet. The env marker alone is not
467
+ // proof — it is inherited — so both budget flags must be present too.
468
+ if (input.inheritedSource === 'budget' &&
469
+ pinnedMb === sizing.oldSpaceMb &&
470
+ semiSpaceMb === sizing.semiSpaceMb) {
471
+ return { respawn: false, sizing };
472
+ }
473
+ const swapWarning = budgetMb > autoCapMb
474
+ ? ` --memory-budget ${budgetMb}MB is above the ${autoCapMb}MB this machine's RAM supports — analyze may swap-thrash.\n`
475
+ : '';
476
+ const replaced = pinnedMb !== null
477
+ ? `the ${pinnedMb}MB --max-old-space-size heap limit`
478
+ : input.autopilotDisabled
479
+ ? `Node's default heap limit`
480
+ : `the auto-sized ${autoCapMb}MB heap cap`;
481
+ return {
482
+ respawn: true,
483
+ sizing,
484
+ warning: ` --memory-budget replaces ${replaced}: re-running analyze with a ${budgetMb}MB heap.\n` +
485
+ swapWarning,
486
+ };
487
+ }
488
+ /** Re-exec the process with the RAM-aware auto heap cap + larger semi-space/stack
489
+ * if we're currently below that, or at the `--memory-budget` heap (#3137).
490
+ *
491
+ * Heap-source precedence (#2649, #3137):
492
+ * - an explicit `--memory-budget` wins over everything, including
493
+ * `GITNEXUS_MEMORY=off` and any `--max-old-space-size` pin;
494
+ * - an explicit per-invocation `--max-old-space-size` (execArgv) wins next;
495
+ * - `GITNEXUS_MEMORY=off` declines the memory autopilot entirely;
496
+ * - an ambient NODE_OPTIONS heap >= the auto cap is honored as-is;
497
+ * - an ambient NODE_OPTIONS heap BELOW the auto cap is treated as an
498
+ * inherited environment default (devcontainers/CI export one for other
499
+ * tooling), not a deliberate per-run choice: warn and respawn with the
500
+ * auto cap. Pre-#2649 this returned early and large repos then OOM'd on
501
+ * whatever heap the environment happened to specify.
502
+ *
503
+ * Paths managed by the auto-sizer or `--memory-budget` record the source in
504
+ * `GITNEXUS_HEAP_LIMIT_SOURCE` (this process when kept, the child env when
505
+ * respawned) so the parse phase's remedy text advises by source. Explicit
506
+ * heap pins and `GITNEXUS_MEMORY=off` intentionally leave it unset. */
438
507
  export async function ensureHeap(options = {}) {
508
+ const nodeOpts = process.env.NODE_OPTIONS || '';
509
+ // The budget is checked BEFORE the GITNEXUS_MEMORY=off return: an explicit
510
+ // flag is a deliberate per-run choice, which the opt-out does not cover.
511
+ if (options.memoryBudget !== undefined) {
512
+ let budgetMb;
513
+ try {
514
+ budgetMb = parseMemoryBudgetMb(options.memoryBudget);
515
+ }
516
+ catch (error) {
517
+ // The CLI's preAction hook rejects bad values first; this covers
518
+ // programmatic callers.
519
+ if (!(error instanceof IntegerOptionError))
520
+ throw error;
521
+ cliError(` ${error.message}\n`);
522
+ process.exitCode = 1;
523
+ return true;
524
+ }
525
+ const decision = resolveBudgetHeap({
526
+ budgetMb,
527
+ execArgv: process.execArgv,
528
+ nodeOptions: nodeOpts,
529
+ autoCapMb: RESPAWN_HEAP_MB,
530
+ autopilotDisabled: memoryAutopilotDisabled(),
531
+ inheritedSource: process.env[HEAP_LIMIT_SOURCE_ENV],
532
+ });
533
+ if (decision.warning)
534
+ cliWarn(decision.warning);
535
+ if (!decision.respawn) {
536
+ process.env[HEAP_LIMIT_SOURCE_ENV] = 'budget';
537
+ return false;
538
+ }
539
+ const { oldSpaceMb, semiSpaceMb } = decision.sizing;
540
+ return respawnWithHeap(`--max-old-space-size=${oldSpaceMb}`, `--max-semi-space-size=${semiSpaceMb}`, 'budget', ` Analysis likely ran out of memory (heap limit set to ${budgetMb}MB by --memory-budget).\n` +
541
+ ` This repository's working set exceeds the budget. Raise it, or omit --memory-budget\n` +
542
+ ` to use the auto-sized cap (a budget above physical RAM causes swap-thrash — use with care):\n` +
543
+ ` gitnexus analyze --memory-budget <MB> [your-args]\n` +
544
+ ` If this persists, it may be a native crash unrelated to heap size.\n`, nodeOpts, options);
545
+ }
439
546
  // Explicit opt-out disables auto-sizing ENTIRELY — both the ambient-pin
440
547
  // override and the default v8-limit respawn — and is honored SILENTLY:
441
548
  // the operator already made the call, and stderr-sensitive consumers
@@ -443,8 +550,7 @@ export async function ensureHeap(options = {}) {
443
550
  // a quiet, single-process run.
444
551
  if (memoryAutopilotDisabled())
445
552
  return false;
446
- const nodeOpts = process.env.NODE_OPTIONS || '';
447
- if (process.execArgv.some((a) => a.startsWith('--max-old-space-size')))
553
+ if (parseMaxOldSpaceMb(process.execArgv.join(' ')) !== null)
448
554
  return false;
449
555
  const ambientHeapMb = parseMaxOldSpaceMb(nodeOpts);
450
556
  if (ambientHeapMb !== null) {
@@ -455,12 +561,23 @@ export async function ensureHeap(options = {}) {
455
561
  }
456
562
  else {
457
563
  const v8Heap = v8.getHeapStatistics().heap_size_limit;
458
- if (v8Heap >= HEAP_MB * 1024 * 1024 * 0.9)
564
+ if (v8Heap >= HEAP_MB * 1024 * 1024 * 0.9) {
565
+ process.env[HEAP_LIMIT_SOURCE_ENV] = 'auto';
459
566
  return false;
567
+ }
460
568
  }
569
+ return respawnWithHeap(HEAP_FLAG, SEMI_FLAG, 'auto', ` Analysis likely ran out of memory (heap cap auto-sized to ${RESPAWN_HEAP_MB}MB ≈ 0.75x RAM).\n` +
570
+ ` This repository's working set exceeds available RAM. Use a machine with more RAM,\n` +
571
+ ` or override the cap (a cap above physical RAM causes swap-thrash — use with care):\n` +
572
+ ` NODE_OPTIONS="--max-old-space-size=<MB>" gitnexus analyze [your-args]\n` +
573
+ ` (Windows: set NODE_OPTIONS=--max-old-space-size=<MB> && gitnexus analyze [your-args])\n` +
574
+ ` If this persists, it may be a native crash unrelated to heap size.\n`, nodeOpts, options);
575
+ }
576
+ /** Run the analyze child with the given heap flags and map its exit onto this process. */
577
+ async function respawnWithHeap(heapFlag, semiFlag, source, oomGuidance, nodeOpts, options) {
461
578
  // --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+, so pass it
462
579
  // only as a direct CLI argument. --max-semi-space-size IS allowed in NODE_OPTIONS.
463
- const cliFlags = [HEAP_FLAG, SEMI_FLAG];
580
+ const cliFlags = [heapFlag, semiFlag];
464
581
  if (!nodeOpts.includes('--stack-size'))
465
582
  cliFlags.push(STACK_FLAG);
466
583
  // Preserve the parent's node flags (execArgv) — dropping them breaks any
@@ -474,7 +591,8 @@ export async function ensureHeap(options = {}) {
474
591
  const childArgs = [...preservedExecArgv, ...cliFlags, ...process.argv.slice(1)];
475
592
  const childEnv = {
476
593
  ...process.env,
477
- NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG} ${SEMI_FLAG}`.trim(),
594
+ NODE_OPTIONS: `${nodeOpts} ${heapFlag} ${semiFlag}`.trim(),
595
+ [HEAP_LIMIT_SOURCE_ENV]: source,
478
596
  };
479
597
  if (shouldBridgeRespawnProgressTty())
480
598
  childEnv[RESPAWN_PROGRESS_ENV] = '1';
@@ -485,12 +603,7 @@ export async function ensureHeap(options = {}) {
485
603
  }
486
604
  if (childExit.status !== 0 || childExit.signal) {
487
605
  if (childProcessLikelyOom(childExit)) {
488
- cliError(` Analysis likely ran out of memory (heap cap auto-sized to ${RESPAWN_HEAP_MB}MB ≈ 0.75x RAM).\n` +
489
- ` This repository's working set exceeds available RAM. Use a machine with more RAM,\n` +
490
- ` or override the cap (a cap above physical RAM causes swap-thrash — use with care):\n` +
491
- ` NODE_OPTIONS="--max-old-space-size=<MB>" gitnexus analyze [your-args]\n` +
492
- ` (Windows: set NODE_OPTIONS=--max-old-space-size=<MB> && gitnexus analyze [your-args])\n` +
493
- ` If this persists, it may be a native crash unrelated to heap size.\n`, { recoveryHint: 'heap-oom-respawn' });
606
+ cliError(oomGuidance, { recoveryHint: 'heap-oom-respawn' });
494
607
  }
495
608
  else if (childProcessLikelyNativeAbort(childExit)) {
496
609
  cliError(` Analysis aborted in a native worker or native binding path.\n` +
@@ -513,6 +626,7 @@ export async function ensureHeap(options = {}) {
513
626
  * for it in the first place.
514
627
  */
515
628
  const ANALYZE_CLI_ENV_KEYS = [
629
+ HEAP_LIMIT_SOURCE_ENV,
516
630
  'GITNEXUS_VERBOSE',
517
631
  'GITNEXUS_PROFILE_DEFERRED',
518
632
  'GITNEXUS_PROFILE_DEFERRED_SLOW_MS',
@@ -563,7 +677,10 @@ const restoreAnalyzeEnv = (snap) => {
563
677
  */
564
678
  export const shouldGenerateCommunitySkillFiles = (options, pipelineResult) => Boolean(options?.skills && pipelineResult && !options?.indexOnly);
565
679
  export const analyzeCommand = async (inputPath, options, runnerIdentityAtBootstrap) => {
566
- if (await ensureHeap())
680
+ // Snapshot before ensureHeap: it records GITNEXUS_HEAP_LIMIT_SOURCE on a
681
+ // kept process, which must not leak into a later programmatic call.
682
+ const envSnap = snapshotAnalyzeEnv();
683
+ if (await ensureHeap({ memoryBudget: options?.memoryBudget }))
567
684
  return;
568
685
  forceHeapOOMForTestIfEnabled();
569
686
  // Install fatal handlers immediately after re-exec resolution so any
@@ -581,7 +698,6 @@ export const analyzeCommand = async (inputPath, options, runnerIdentityAtBootstr
581
698
  // exiting, restoration is moot. For early-return paths (validation
582
699
  // errors) and the alreadyUpToDate fast path the finally restores the
583
700
  // pre-call values.
584
- const envSnap = snapshotAnalyzeEnv();
585
701
  try {
586
702
  await analyzeCommandImpl(inputPath, options, runnerIdentityAtBootstrap);
587
703
  }
@@ -744,15 +860,18 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
744
860
  // previous call leaked.
745
861
  let workerPoolSize;
746
862
  if (options.workers !== undefined) {
747
- const parsedWorkers = Number(options.workers);
748
- if (!Number.isInteger(parsedWorkers) || parsedWorkers < 1) {
863
+ try {
864
+ workerPoolSize = parseIntegerOption(options.workers, '--workers', { minimum: 1 });
865
+ }
866
+ catch (error) {
867
+ if (!(error instanceof IntegerOptionError))
868
+ throw error;
749
869
  cliError(' --workers must be a positive integer (>= 1). ' +
750
870
  'GitNexus parses through a worker pool only — there is no sequential ' +
751
871
  'mode, so 0 is not allowed. Omit --workers for an auto-sized pool.\n');
752
872
  process.exitCode = 1;
753
873
  return;
754
874
  }
755
- workerPoolSize = parsedWorkers;
756
875
  }
757
876
  const processDetectionFromFlags = parseProcessDetectionBudgetStrings({
758
877
  maxProcesses: options.maxProcesses,
@@ -768,26 +887,34 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
768
887
  // process.exit() leaves the progress bar's hidden cursor uncleared).
769
888
  let embeddingsNodeLimit;
770
889
  if (typeof options.embeddings === 'string') {
771
- const parsed = Number(options.embeddings);
772
- if (!Number.isInteger(parsed) || parsed < 0) {
890
+ try {
891
+ embeddingsNodeLimit = parseIntegerOption(options.embeddings, '--embeddings', {
892
+ minimum: 0,
893
+ });
894
+ }
895
+ catch (error) {
896
+ if (!(error instanceof IntegerOptionError))
897
+ throw error;
773
898
  cliError(` --embeddings expects a non-negative integer (got "${options.embeddings}"). ` +
774
899
  `Pass 0 to disable the safety cap, or omit the value to keep the default.\n`);
775
900
  process.exitCode = 1;
776
901
  return;
777
902
  }
778
- embeddingsNodeLimit = parsed;
779
903
  }
780
904
  const embeddingsEnabled = !!options.embeddings;
781
905
  const setPositiveEnv = (optionName, envName, value) => {
782
906
  if (value === undefined)
783
907
  return true;
784
- const parsed = Number(value);
785
- if (!Number.isInteger(parsed) || parsed <= 0) {
786
- cliError(` ${optionName} must be a positive integer.\n`);
908
+ try {
909
+ process.env[envName] = String(parseIntegerOption(value, optionName, { minimum: 1 }));
910
+ }
911
+ catch (error) {
912
+ if (!(error instanceof IntegerOptionError))
913
+ throw error;
914
+ cliError(` ${error.message}.\n`);
787
915
  process.exitCode = 1;
788
916
  return false;
789
917
  }
790
- process.env[envName] = String(parsed);
791
918
  return true;
792
919
  };
793
920
  if (!setPositiveEnv('--embedding-threads', 'GITNEXUS_EMBEDDING_THREADS', options.embeddingThreads) ||
@@ -955,7 +1082,7 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
955
1082
  // injection, including community skill writes that `--skills` would normally
956
1083
  // produce. Surface the override explicitly so users don't wonder why a
957
1084
  // pipeline re-index ran but no skill files appeared. The pipeline still
958
- // re-runs (see `force: options.force || options.skills` below); the warning
1085
+ // re-runs (`skills` is passed to runFullAnalysis below); the warning
959
1086
  // is purely about the dropped post-index write step.
960
1087
  if (options.indexOnly && options.skills) {
961
1088
  console.log(' Note: --index-only overrides --skills; community skill files will not be written.\n');
@@ -1093,10 +1220,12 @@ const analyzeCommandImpl = async (inputPath, cliOptions, runnerIdentityAtBootstr
1093
1220
  const skipAgentsMd = skipAll || options.skipAgentsMd;
1094
1221
  const skipSkills = skipAll || options.skipSkills;
1095
1222
  const runOptions = {
1096
- // Pipeline re-index — OR'd with --skills because skill generation needs
1097
- // a fresh pipelineResult, and with --no-parse-cache because bypassing
1098
- // parser output is meaningful only when the pipeline runs.
1099
- force: options.force || options.skills || options.parseCache === false,
1223
+ // The user's own --force only. --skills (skill generation needs a fresh
1224
+ // pipelineResult) and --no-parse-cache (bypassing parser output means
1225
+ // anything only when the pipeline runs) each force the rebuild inside
1226
+ // runFullAnalysis under their own named reason (#3137).
1227
+ force: options.force,
1228
+ skills: options.skills,
1100
1229
  useParseCache: options.parseCache !== false,
1101
1230
  repairFts: options.repairFts,
1102
1231
  skipFts: options.skipFts,
@@ -69,6 +69,7 @@ const OPTION_DESCRIPTION_KEYS = {
69
69
  'analyze|--max-file-size <kb>': 'help.option.analyze.maxFileSize',
70
70
  'analyze|--worker-timeout <seconds>': 'help.option.analyze.workerTimeout',
71
71
  'analyze|--wal-checkpoint-threshold <bytes>': 'help.option.analyze.walCheckpointThreshold',
72
+ 'analyze|--memory-budget <mb>': 'help.option.analyze.memoryBudget',
72
73
  'analyze|--workers <n>': 'help.option.analyze.workers',
73
74
  'analyze|--max-processes <n>': 'help.option.analyze.maxProcesses',
74
75
  'analyze|--max-process-branching <n>': 'help.option.analyze.maxProcessBranching',
@@ -229,6 +229,7 @@ export declare const en: {
229
229
  readonly 'help.option.analyze.maxFileSize': 'Skip files larger than this (KB). Default: 512. Hard cap: 32768 (tree-sitter limit).';
230
230
  readonly 'help.option.analyze.workerTimeout': 'Worker sub-batch idle timeout before retry/fallback. Default: 30.';
231
231
  readonly 'help.option.analyze.walCheckpointThreshold': 'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).';
232
+ readonly 'help.option.analyze.memoryBudget': 'Main-thread V8 heap size in MB for analyze (integer >= 200). Re-runs analyze with exactly this heap, overriding the RAM/cgroup auto-sizer and any --max-old-space-size pin; parse workers keep their own heap caps.';
232
233
  readonly 'help.option.analyze.workers': 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.';
233
234
  readonly 'help.option.analyze.maxProcesses': 'Process-detection process cap (positive integer). Replaces the dynamic max(20, round(symbols/10)) formula. Default: dynamic.';
234
235
  readonly 'help.option.analyze.maxProcessBranching': 'Process-detection per-node branching cap (positive integer). Default: 4.';
@@ -231,6 +231,7 @@ export const en = {
231
231
  'help.option.analyze.maxFileSize': 'Skip files larger than this (KB). Default: 512. Hard cap: 32768 (tree-sitter limit).',
232
232
  'help.option.analyze.workerTimeout': 'Worker sub-batch idle timeout before retry/fallback. Default: 30.',
233
233
  'help.option.analyze.walCheckpointThreshold': 'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).',
234
+ 'help.option.analyze.memoryBudget': 'Main-thread V8 heap size in MB for analyze (integer >= 200). Re-runs analyze with exactly this heap, overriding the RAM/cgroup auto-sizer and any --max-old-space-size pin; parse workers keep their own heap caps.',
234
235
  'help.option.analyze.workers': 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
235
236
  'help.option.analyze.maxProcesses': 'Process-detection process cap (positive integer). Replaces the dynamic max(20, round(symbols/10)) formula. Default: dynamic.',
236
237
  'help.option.analyze.maxProcessBranching': 'Process-detection per-node branching cap (positive integer). Default: 4.',
@@ -230,6 +230,7 @@ export declare const cliResources: {
230
230
  readonly 'help.option.analyze.maxFileSize': 'Skip files larger than this (KB). Default: 512. Hard cap: 32768 (tree-sitter limit).';
231
231
  readonly 'help.option.analyze.workerTimeout': 'Worker sub-batch idle timeout before retry/fallback. Default: 30.';
232
232
  readonly 'help.option.analyze.walCheckpointThreshold': 'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).';
233
+ readonly 'help.option.analyze.memoryBudget': 'Main-thread V8 heap size in MB for analyze (integer >= 200). Re-runs analyze with exactly this heap, overriding the RAM/cgroup auto-sizer and any --max-old-space-size pin; parse workers keep their own heap caps.';
233
234
  readonly 'help.option.analyze.workers': 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.';
234
235
  readonly 'help.option.analyze.maxProcesses': 'Process-detection process cap (positive integer). Replaces the dynamic max(20, round(symbols/10)) formula. Default: dynamic.';
235
236
  readonly 'help.option.analyze.maxProcessBranching': 'Process-detection per-node branching cap (positive integer). Default: 4.';
@@ -556,6 +557,7 @@ export declare const cliResources: {
556
557
  'help.option.analyze.maxFileSize': string;
557
558
  'help.option.analyze.workerTimeout': string;
558
559
  'help.option.analyze.walCheckpointThreshold': string;
560
+ 'help.option.analyze.memoryBudget': string;
559
561
  'help.option.analyze.workers': string;
560
562
  'help.option.analyze.maxProcesses': string;
561
563
  'help.option.analyze.maxProcessBranching': string;
@@ -229,6 +229,7 @@ export declare const zhCN: {
229
229
  'help.option.analyze.maxFileSize': string;
230
230
  'help.option.analyze.workerTimeout': string;
231
231
  'help.option.analyze.walCheckpointThreshold': string;
232
+ 'help.option.analyze.memoryBudget': string;
232
233
  'help.option.analyze.workers': string;
233
234
  'help.option.analyze.maxProcesses': string;
234
235
  'help.option.analyze.maxProcessBranching': string;
@@ -229,6 +229,7 @@ export const zhCN = {
229
229
  'help.option.analyze.maxFileSize': '跳过大于该值的文件(KB)。默认:512。硬上限:32768(tree-sitter 限制)。',
230
230
  'help.option.analyze.workerTimeout': 'Worker 子批次空闲超时,超时后重试/回退。默认:30。',
231
231
  'help.option.analyze.walCheckpointThreshold': 'analyze 期间 LadybugDB WAL 自动 checkpoint 阈值(字节,整数 >= -1;默认:67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。',
232
+ 'help.option.analyze.memoryBudget': 'analyze 主线程 V8 堆大小(MB,整数 >= 200)。以该堆大小重新运行 analyze,覆盖按 RAM/cgroup 自动计算的上限及任何 --max-old-space-size 设置;解析 worker 各自保留独立的堆上限。',
232
233
  'help.option.analyze.workers': '解析 worker 池大小(>=1)。默认:cores-1,最多 16,按仓库规模自适应。',
233
234
  'help.option.analyze.maxProcesses': '流程检测的流程数量上限(正整数)。覆盖动态的 max(20, round(symbols/10)) 公式。默认:动态。',
234
235
  'help.option.analyze.maxProcessBranching': '流程检测的单节点分支上限(正整数)。默认:4。',