gitnexus 1.6.10-rc.89 → 1.6.10-rc.90
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli/doctor.d.ts +11 -0
- package/dist/cli/doctor.js +19 -1
- package/dist/core/lbug/lbug-adapter.js +19 -2
- package/dist/core/lbug/lbug-config.d.ts +70 -9
- package/dist/core/lbug/lbug-config.js +133 -18
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -484,7 +484,7 @@ Configure the behavior with these environment variables:
|
|
|
484
484
|
| `GITNEXUS_FTS_CJK_SEGMENTATION` | `none`, `bigram` | `none` | `bigram` inserts overlapping character-bigram boundaries into Chinese/Japanese Han-ideograph spans in `content`/`description` before FTS indexing, so LadybugDB's space-only tokenizer can see sub-phrase word boundaries. Scoped to CJK Unified Ideographs only — Japanese Hiragana/Katakana and Korean Hangul are not currently segmented. Unlike `GITNEXUS_FTS_STEMMER`, this rewrites stored text — enabling it on an already-indexed repo requires a full `gitnexus analyze --force`; neither `--repair-fts` nor a plain incremental `analyze` applies it to previously-indexed files. Set the same value wherever `analyze` and search-serving processes (CLI query, MCP server, web server) run. |
|
|
485
485
|
| `GITNEXUS_COMMUNITY_ENGINE` | `graphology`, `icebug`, `auto` | `graphology` | Community-detection engine used during analyze. `graphology` uses the bundled default path. `icebug` and `auto` currently behave identically: both try the experimental Icebug CSR path and fall back to Graphology if the optional native module is unavailable or incompatible. |
|
|
486
486
|
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | integer `>= -1` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold during analyze (bytes). Auto-checkpoint remains enabled; `-1` keeps Ladybug's stock ~16 MiB. Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. |
|
|
487
|
-
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. |
|
|
487
|
+
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. During `analyze` the pool is right-sized to the graph and, on non-4 KiB-page hosts (Apple Silicon 16 KiB, Ascend/aarch64 64 KiB), scaled by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. |
|
|
488
488
|
| `GITNEXUS_LBUG_MAX_DB_SIZE` | positive integer (bytes) | `17179869184` (16 GiB) | Upper bound for a single LadybugDB database file. This is an mmap/disk-address-space ceiling, not a memory limit — it does not constrain the buffer pool (use `GITNEXUS_LBUG_BUFFER_POOL_SIZE` for that). Raise it when indexing genuinely huge monorepos; invalid values silently fall back to the default. |
|
|
489
489
|
|
|
490
490
|
```bash
|
package/dist/cli/doctor.d.ts
CHANGED
|
@@ -36,4 +36,15 @@ export declare function localEmbeddingDoctorStatus(opts: {
|
|
|
36
36
|
* detect the OS page size at runtime).
|
|
37
37
|
*/
|
|
38
38
|
export declare function pageSizeDoctorLines(pageSize: number | undefined, ladybugVersion: string | undefined): string[];
|
|
39
|
+
/**
|
|
40
|
+
* The hintless buffer-pool doctor line (#2631) — the pool the next Database
|
|
41
|
+
* open in THIS process would get. Same plain-params testable-helper shape as
|
|
42
|
+
* pageSizeDoctorLines above. `pool` is getEffectiveBufferPoolSize(): `0` is
|
|
43
|
+
* the pass-through sentinel for LadybugDB's native 80%-of-RAM default, never
|
|
44
|
+
* printed as "0 MiB". `envRaw` (the raw GITNEXUS_LBUG_BUFFER_POOL_SIZE value)
|
|
45
|
+
* marks operator-supplied absolute values as "(env override)" — no scaling
|
|
46
|
+
* suffix: the hintless default is deliberately unscaled (#2557), and an env
|
|
47
|
+
* value is absolute, so a "×N" note would misdescribe both.
|
|
48
|
+
*/
|
|
49
|
+
export declare function poolSizeDoctorLine(pool: number, envRaw: string | undefined): string;
|
|
39
50
|
export declare const doctorCommand: () => Promise<void>;
|
package/dist/cli/doctor.js
CHANGED
|
@@ -5,7 +5,7 @@ import { getLocalEmbeddingRuntimeBlocker, localEmbeddingPrefixUnloadableMessage,
|
|
|
5
5
|
import { isPrefixRuntimeLoadable, resolveEmbeddingRuntime, } from '../core/embeddings/runtime-install.js';
|
|
6
6
|
import { cudaRedirectDoctorStatus } from '../core/embeddings/onnxruntime-node-resolver.js';
|
|
7
7
|
import { checkLbugNative, probeFtsExtensionLoad, probeVectorExtensionLoad, } from '../core/lbug/native-check.js';
|
|
8
|
-
import { getOsPageSize, isPageSizeAwareLadybug } from '../core/lbug/lbug-config.js';
|
|
8
|
+
import { getEffectiveBufferPoolSize, getOsPageSize, isPageSizeAwareLadybug, } from '../core/lbug/lbug-config.js';
|
|
9
9
|
import { diagnoseExtensionLoad } from '../core/lbug/extension-load-error.js';
|
|
10
10
|
import { getExtensionInstallPolicy } from '../core/lbug/extension-loader.js';
|
|
11
11
|
import { t } from './i18n/index.js';
|
|
@@ -115,6 +115,21 @@ export function pageSizeDoctorLines(pageSize, ladybugVersion) {
|
|
|
115
115
|
}
|
|
116
116
|
return lines;
|
|
117
117
|
}
|
|
118
|
+
/**
|
|
119
|
+
* The hintless buffer-pool doctor line (#2631) — the pool the next Database
|
|
120
|
+
* open in THIS process would get. Same plain-params testable-helper shape as
|
|
121
|
+
* pageSizeDoctorLines above. `pool` is getEffectiveBufferPoolSize(): `0` is
|
|
122
|
+
* the pass-through sentinel for LadybugDB's native 80%-of-RAM default, never
|
|
123
|
+
* printed as "0 MiB". `envRaw` (the raw GITNEXUS_LBUG_BUFFER_POOL_SIZE value)
|
|
124
|
+
* marks operator-supplied absolute values as "(env override)" — no scaling
|
|
125
|
+
* suffix: the hintless default is deliberately unscaled (#2557), and an env
|
|
126
|
+
* value is absolute, so a "×N" note would misdescribe both.
|
|
127
|
+
*/
|
|
128
|
+
export function poolSizeDoctorLine(pool, envRaw) {
|
|
129
|
+
const value = pool === 0 ? 'native 80% of RAM' : `${Math.round(pool / (1024 * 1024))} MiB`;
|
|
130
|
+
const envNote = envRaw !== undefined && envRaw.trim().length > 0 ? ' (env override)' : '';
|
|
131
|
+
return ` ${padDisplayEnd('pool size', 10)}${value}${envNote}`;
|
|
132
|
+
}
|
|
118
133
|
export const doctorCommand = async () => {
|
|
119
134
|
const fingerprint = getRuntimeFingerprint();
|
|
120
135
|
const capabilities = getRuntimeCapabilities();
|
|
@@ -132,6 +147,9 @@ export const doctorCommand = async () => {
|
|
|
132
147
|
for (const line of pageSizeDoctorLines(getOsPageSize(), fingerprint.ladybugdb)) {
|
|
133
148
|
console.log(line);
|
|
134
149
|
}
|
|
150
|
+
// Hintless buffer pool for the next DB open (#2631). Literal label like
|
|
151
|
+
// the page size line above (no i18n key).
|
|
152
|
+
console.log(poolSizeDoctorLine(getEffectiveBufferPoolSize(), process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE));
|
|
135
153
|
const nativeCheck = checkLbugNative();
|
|
136
154
|
if (nativeCheck.ok) {
|
|
137
155
|
console.log(` ${padDisplayEnd('native', 10)}✓ lbugjs.node loaded`);
|
|
@@ -14,7 +14,7 @@ import { streamAllCSVsToDisk } from './csv-generator.js';
|
|
|
14
14
|
import { getNodeLabel as deriveNodeLabel } from './rel-pair-routing.js';
|
|
15
15
|
import { EMBEDDABLE_LABELS } from '../embeddings/types.js';
|
|
16
16
|
import { extensionManager, resolveAnalyzeInstallPolicy, } from './extension-loader.js';
|
|
17
|
-
import { classifyDeleteAllError, closeLbugConnection, HANDLE_RELEASE_PROBE_ATTEMPTS, HANDLE_RELEASE_PROBE_DELAY_MS, isDbBusyError, isOpenRetryExhausted, isWalCorruptionError, openLbugConnection, sleep, toNativeSafePath, resolveNativeSafeStorageDir, WAL_RECOVERY_SUGGESTION, waitForWindowsHandleRelease, } from './lbug-config.js';
|
|
17
|
+
import { classifyDeleteAllError, closeLbugConnection, HANDLE_RELEASE_PROBE_ATTEMPTS, HANDLE_RELEASE_PROBE_DELAY_MS, isDbBusyError, isOpenRetryExhausted, isWalCorruptionError, bufferPoolExhaustionRemedy, openLbugConnection, sleep, toNativeSafePath, resolveNativeSafeStorageDir, WAL_RECOVERY_SUGGESTION, waitForWindowsHandleRelease, } from './lbug-config.js';
|
|
18
18
|
import { finalizeLbugSidecarsAfterClose, guardWalQuarantine, isMissingShadowSidecarError, isReadOnlyShadowReplayError, lbugLockRemediation, preflightLbugSidecars, quarantineWalForMissingShadow, renameFailureMessage, shadowSidecarRecoveryMessage, } from './sidecar-recovery.js';
|
|
19
19
|
import { logger } from '../logger.js';
|
|
20
20
|
/**
|
|
@@ -796,7 +796,12 @@ const copyNodeCSVs = async (targetConn, nodeFileEntries, log, totalSteps) => {
|
|
|
796
796
|
const copyQuery = getCopyQuery(table, normalizeCopyPath(csvPath));
|
|
797
797
|
await copyCsvWithRetry(targetConn, copyQuery, (retryErr) => {
|
|
798
798
|
const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr);
|
|
799
|
-
|
|
799
|
+
// Pool exhaustion gets a remedy (#2631): the raw binder text gives the
|
|
800
|
+
// operator nothing to act on, and on non-4K-page hosts (Ascend aarch64,
|
|
801
|
+
// Apple Silicon) the pool bills up to pageSize/4KiB x faster than the
|
|
802
|
+
// sizing was calibrated for — name the knob and the mechanism.
|
|
803
|
+
const remedy = bufferPoolExhaustionRemedy(retryMsg);
|
|
804
|
+
throw new Error(`COPY failed for ${table}: ${retryMsg.slice(0, 200)}${remedy ? ` ${remedy}` : ''}`);
|
|
800
805
|
});
|
|
801
806
|
}
|
|
802
807
|
};
|
|
@@ -944,6 +949,7 @@ pdgEmitManifest) => {
|
|
|
944
949
|
let tFallback = tCopyNodes;
|
|
945
950
|
const insertedRels = totalValidRels;
|
|
946
951
|
const warnings = [];
|
|
952
|
+
let poolRemedyIssued = false;
|
|
947
953
|
if (insertedRels > 0) {
|
|
948
954
|
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`);
|
|
949
955
|
let pairIdx = 0;
|
|
@@ -966,6 +972,17 @@ pdgEmitManifest) => {
|
|
|
966
972
|
await copyCsvWithRetry(writeConn, copyQuery, (retryErr) => {
|
|
967
973
|
const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr);
|
|
968
974
|
warnings.push(`${fromLabel}->${toLabel} (${rows} edges): ${retryMsg.slice(0, 80)}`);
|
|
975
|
+
// One remedy per bulk load, not per pair (#2631): pool exhaustion
|
|
976
|
+
// repeats for every remaining pair once it starts. logger.warn, not
|
|
977
|
+
// just warnings.push — the returned warnings array has no consumer at
|
|
978
|
+
// any call site, so a push alone would leave the remedy invisible
|
|
979
|
+
// while the row-by-row fallback quietly degrades the load.
|
|
980
|
+
const remedy = poolRemedyIssued ? undefined : bufferPoolExhaustionRemedy(retryMsg);
|
|
981
|
+
if (remedy) {
|
|
982
|
+
poolRemedyIssued = true;
|
|
983
|
+
warnings.push(remedy);
|
|
984
|
+
logger.warn(remedy);
|
|
985
|
+
}
|
|
969
986
|
failedPairEdges += rows;
|
|
970
987
|
failedPairCsvPaths.add(pairCsvPath);
|
|
971
988
|
});
|
|
@@ -61,16 +61,68 @@ export declare function resolveNativeSafeStorageDir(storagePath: string, subdir:
|
|
|
61
61
|
*/
|
|
62
62
|
export declare const LBUG_MAX_DB_SIZE: number;
|
|
63
63
|
export declare const parseWalCheckpointThreshold: (raw: string | undefined) => number | undefined;
|
|
64
|
+
/**
|
|
65
|
+
* How much the OS page size amplifies buffer-pool consumption (#2631).
|
|
66
|
+
*
|
|
67
|
+
* LadybugDB's VM region charges pool budget per DISCARD GRANULE, not per
|
|
68
|
+
* frame: `discardGranuleSize = max(frameSize, osPageSize)` (vm_region.cpp),
|
|
69
|
+
* `claimFrame` bills the whole granule when its first 4 KiB frame becomes
|
|
70
|
+
* resident, and `releaseFrame` refunds only when the granule's LAST frame
|
|
71
|
+
* leaves. On a 64 KiB-page kernel (aarch64 openEuler — Ascend hosts) that is
|
|
72
|
+
* 16 frames per granule: scattered access is billed up to 16× its real bytes,
|
|
73
|
+
* and whole eviction passes can evict frames yet refund nothing — which is
|
|
74
|
+
* exactly the engine's "buffer pool is full and no memory could be freed"
|
|
75
|
+
* throw. Apple Silicon macOS (16 KiB pages) is the same mechanism at 4×.
|
|
76
|
+
*
|
|
77
|
+
* So the ANALYZE-path pool sizes (the per-element estimate, the COPY-safety
|
|
78
|
+
* floor, and the cap the hint is clamped against) are scaled by this ratio:
|
|
79
|
+
* the budget must cover worst-case granule charging or COPY dies on non-4K
|
|
80
|
+
* hosts with a pool that would be ample on x86. The hintless default
|
|
81
|
+
* (defaultBufferPoolSize — MCP serve, doctor, native-check) is deliberately
|
|
82
|
+
* NOT scaled: the pool is a native eager allocation committed at DB open
|
|
83
|
+
* (measured — see POOL_BYTES_PER_ELEMENT below), so scaling the global
|
|
84
|
+
* default would revert the #2557 OOM cap on every 16 KiB/64 KiB host. If the
|
|
85
|
+
* engine ever charges per-frame (or ships page-size-matched frames), this
|
|
86
|
+
* collapses back to 1 and the scaling disappears.
|
|
87
|
+
*
|
|
88
|
+
* Fail-safe: an undetectable page size (win32 — where the granule mechanism
|
|
89
|
+
* is absent anyway — or a failed `getconf`) means ratio 1, i.e. today's
|
|
90
|
+
* behavior.
|
|
91
|
+
*/
|
|
92
|
+
export declare const granuleRatio: (pageSize?: number | undefined) => number;
|
|
64
93
|
/**
|
|
65
94
|
* Size the buffer pool to an estimated graph size (node + relationship count),
|
|
66
|
-
* clamped to [ADAPTIVE_POOL_FLOOR,
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
-
*
|
|
95
|
+
* clamped to [ADAPTIVE_POOL_FLOOR, scaledAnalyzePoolCap], with every term
|
|
96
|
+
* scaled by granuleRatio (#2631): on non-4K hosts the engine bills pool
|
|
97
|
+
* budget per OS-page-sized granule, so the same graph consumes up to
|
|
98
|
+
* pageSize/4096 × the budget it needs on x86. On 4 KiB hosts the ratio is 1
|
|
99
|
+
* and this is byte-identical to the pre-#2631 behavior. The estimate is never
|
|
100
|
+
* above the page-size-scaled #2557 cap bounded by 80% of RAM, never below the
|
|
101
|
+
* scaled COPY-safety floor; the hintless default stays unscaled.
|
|
102
|
+
*
|
|
103
|
+
* `pageSize` is a test seam (the pageSizeDoctorLines convention); production
|
|
104
|
+
* callers omit it and get the memoized real OS page size.
|
|
70
105
|
*/
|
|
71
|
-
export declare const estimateBufferPool: (graphElementCount: number) => number;
|
|
106
|
+
export declare const estimateBufferPool: (graphElementCount: number, pageSize?: number | undefined) => number;
|
|
72
107
|
/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */
|
|
73
108
|
export declare const setBufferPoolSizeHint: (bytes: number | undefined) => void;
|
|
109
|
+
/**
|
|
110
|
+
* Doctor-facing view of the pool size the next Database open would get
|
|
111
|
+
* (#2631): env override > clamped hint > unscaled hintless default. Read-only;
|
|
112
|
+
* doctor prints it next to the page-size lines so support triage sees the
|
|
113
|
+
* sizing inputs at a glance. `0` is the pass-through sentinel for LadybugDB's
|
|
114
|
+
* native 80%-of-RAM default — callers must label it, not print "0 MiB".
|
|
115
|
+
*/
|
|
116
|
+
export declare const getEffectiveBufferPoolSize: () => number;
|
|
117
|
+
/**
|
|
118
|
+
* Actionable remedy for a buffer-pool exhaustion error (#2631), or undefined
|
|
119
|
+
* when `message` is not that class. Cause → consequence → remedy, the
|
|
120
|
+
* diagnoseExtensionLoad convention: names the effective pool, the override
|
|
121
|
+
* knob, and — on non-4K hosts — the granule amplification that makes the
|
|
122
|
+
* budget exhaust early (the reporter's Ascend/aarch64 64 KiB kernel billed a
|
|
123
|
+
* pool up to 16× faster than the same analyze on x86).
|
|
124
|
+
*/
|
|
125
|
+
export declare const bufferPoolExhaustionRemedy: (message: string, pageSize?: number | undefined) => string | undefined;
|
|
74
126
|
export declare const WAL_RECOVERY_SUGGESTION = "WAL corruption detected. Run `gitnexus analyze --force` to rebuild the index.";
|
|
75
127
|
export declare function isWalCorruptionError(err: unknown): boolean;
|
|
76
128
|
/**
|
|
@@ -83,8 +135,12 @@ export declare const isLbugCheckpointIoError: (err: unknown) => boolean;
|
|
|
83
135
|
* True when `err` looks like the LadybugDB buffer manager failing to release
|
|
84
136
|
* frame memory — the failure mode of a 4 KiB page-size assumption on a
|
|
85
137
|
* 16 KiB/64 KiB-page kernel (#1231). Deliberately does NOT match the
|
|
86
|
-
* generic "buffer pool is full" exhaustion error
|
|
87
|
-
* problem
|
|
138
|
+
* generic "buffer pool is full" exhaustion error: that one is handled as a
|
|
139
|
+
* SIZING problem — though since #2631 we know page size drives sizing too
|
|
140
|
+
* (the engine bills pool budget per OS-page-sized discard granule, so non-4K
|
|
141
|
+
* hosts exhaust the same budget up to pageSize/4096× earlier; see
|
|
142
|
+
* granuleRatio, which scales the pool accordingly, and
|
|
143
|
+
* bufferPoolExhaustionRemedy, which explains it to the operator).
|
|
88
144
|
*/
|
|
89
145
|
export declare const isLbugPageSizeFrameError: (err: unknown) => boolean;
|
|
90
146
|
/**
|
|
@@ -94,6 +150,13 @@ export declare const isLbugPageSizeFrameError: (err: unknown) => boolean;
|
|
|
94
150
|
* side of showing the upgrade hint.
|
|
95
151
|
*/
|
|
96
152
|
export declare const isPageSizeAwareLadybug: (version: string | undefined) => boolean;
|
|
153
|
+
/**
|
|
154
|
+
* Test seam (the `_captureLogger` convention): pin the memoized OS page size
|
|
155
|
+
* so sizing tests are host-independent — without this they would silently
|
|
156
|
+
* drift on 16 KiB-page Apple Silicon runners. `number` pins a value, `null`
|
|
157
|
+
* pins "undetectable", `undefined` clears the memo so the next call re-probes.
|
|
158
|
+
*/
|
|
159
|
+
export declare const _setOsPageSizeForTests: (pageSize: number | null | undefined) => void;
|
|
97
160
|
/**
|
|
98
161
|
* OS memory page size in bytes, or `undefined` when it cannot be determined
|
|
99
162
|
* (Windows, missing getconf, sandboxed exec). Node exposes no page-size API,
|
|
@@ -102,8 +165,6 @@ export declare const isPageSizeAwareLadybug: (version: string | undefined) => bo
|
|
|
102
165
|
* an explicit killSignal (see the options comment below).
|
|
103
166
|
*/
|
|
104
167
|
export declare const getOsPageSize: () => number | undefined;
|
|
105
|
-
/** Exported only for unit tests — clears the getconf probe cache. */
|
|
106
|
-
export declare const _resetOsPageSizeCacheForTest: () => void;
|
|
107
168
|
type LbugModule = typeof lbug;
|
|
108
169
|
export interface LbugDatabaseOptions {
|
|
109
170
|
readOnly?: boolean;
|
|
@@ -327,15 +327,72 @@ const parseBufferPoolSize = (raw) => {
|
|
|
327
327
|
return undefined;
|
|
328
328
|
return Math.floor(parsed);
|
|
329
329
|
};
|
|
330
|
+
/**
|
|
331
|
+
* The buffer-manager frame size compiled into every shipped `@ladybugdb/core`
|
|
332
|
+
* binary (`LBUG_PAGE_SIZE_LOG2 = 12` in the engine's CMake) — frames are 4 KiB
|
|
333
|
+
* on every platform, independent of the OS page size.
|
|
334
|
+
*/
|
|
335
|
+
const LBUG_ASSUMED_FRAME_SIZE = 4096;
|
|
336
|
+
/**
|
|
337
|
+
* How much the OS page size amplifies buffer-pool consumption (#2631).
|
|
338
|
+
*
|
|
339
|
+
* LadybugDB's VM region charges pool budget per DISCARD GRANULE, not per
|
|
340
|
+
* frame: `discardGranuleSize = max(frameSize, osPageSize)` (vm_region.cpp),
|
|
341
|
+
* `claimFrame` bills the whole granule when its first 4 KiB frame becomes
|
|
342
|
+
* resident, and `releaseFrame` refunds only when the granule's LAST frame
|
|
343
|
+
* leaves. On a 64 KiB-page kernel (aarch64 openEuler — Ascend hosts) that is
|
|
344
|
+
* 16 frames per granule: scattered access is billed up to 16× its real bytes,
|
|
345
|
+
* and whole eviction passes can evict frames yet refund nothing — which is
|
|
346
|
+
* exactly the engine's "buffer pool is full and no memory could be freed"
|
|
347
|
+
* throw. Apple Silicon macOS (16 KiB pages) is the same mechanism at 4×.
|
|
348
|
+
*
|
|
349
|
+
* So the ANALYZE-path pool sizes (the per-element estimate, the COPY-safety
|
|
350
|
+
* floor, and the cap the hint is clamped against) are scaled by this ratio:
|
|
351
|
+
* the budget must cover worst-case granule charging or COPY dies on non-4K
|
|
352
|
+
* hosts with a pool that would be ample on x86. The hintless default
|
|
353
|
+
* (defaultBufferPoolSize — MCP serve, doctor, native-check) is deliberately
|
|
354
|
+
* NOT scaled: the pool is a native eager allocation committed at DB open
|
|
355
|
+
* (measured — see POOL_BYTES_PER_ELEMENT below), so scaling the global
|
|
356
|
+
* default would revert the #2557 OOM cap on every 16 KiB/64 KiB host. If the
|
|
357
|
+
* engine ever charges per-frame (or ships page-size-matched frames), this
|
|
358
|
+
* collapses back to 1 and the scaling disappears.
|
|
359
|
+
*
|
|
360
|
+
* Fail-safe: an undetectable page size (win32 — where the granule mechanism
|
|
361
|
+
* is absent anyway — or a failed `getconf`) means ratio 1, i.e. today's
|
|
362
|
+
* behavior.
|
|
363
|
+
*/
|
|
364
|
+
export const granuleRatio = (pageSize = getOsPageSize()) => {
|
|
365
|
+
if (pageSize === undefined || !Number.isFinite(pageSize))
|
|
366
|
+
return 1;
|
|
367
|
+
return Math.max(1, Math.floor(pageSize / LBUG_ASSUMED_FRAME_SIZE));
|
|
368
|
+
};
|
|
369
|
+
/**
|
|
370
|
+
* Hintless pool default — MCP serve, doctor, native-check, any open without a
|
|
371
|
+
* per-run hint. Deliberately UNSCALED (#2557): the pool is an eager native
|
|
372
|
+
* allocation at DB open, so a page-size-scaled default would hand a
|
|
373
|
+
* long-lived `gitnexus mcp` on a 16 KiB/64 KiB host up to 80% of RAM — the
|
|
374
|
+
* exact OOM exposure the 2 GiB cap was added to remove.
|
|
375
|
+
*/
|
|
330
376
|
const defaultBufferPoolSize = () => Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)));
|
|
331
377
|
/**
|
|
332
|
-
*
|
|
333
|
-
*
|
|
334
|
-
*
|
|
335
|
-
*
|
|
336
|
-
*
|
|
378
|
+
* Upper bound for the ANALYZE-path (hinted) pool: the #2557 cap scaled by the
|
|
379
|
+
* granule ratio, still bounded by 80% of RAM. Scaling only this bound — and
|
|
380
|
+
* not defaultBufferPoolSize — is what lets the #2631 fix take effect during
|
|
381
|
+
* the bulk COPY without touching hintless opens: with an unscaled cap the
|
|
382
|
+
* min() below would clamp the scaled COPY floor straight back to 2 GiB.
|
|
337
383
|
*/
|
|
338
|
-
const
|
|
384
|
+
const scaledAnalyzePoolCap = (pageSize) => Math.min(DEFAULT_BUFFER_POOL_CAP * granuleRatio(pageSize), Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)));
|
|
385
|
+
/**
|
|
386
|
+
* Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR × granuleRatio,
|
|
387
|
+
* scaledAnalyzePoolCap]. The lower bound keeps LadybugDB's COPY viable
|
|
388
|
+
* (scaled because the granule accounting inflates consumption on non-4K
|
|
389
|
+
* hosts, see granuleRatio); the upper bound means the hint can never exceed
|
|
390
|
+
* the page-size-scaled #2557 cap or 80% of RAM — and on a machine whose cap
|
|
391
|
+
* is below the COPY floor, the cap wins, so the pool is never over-committed.
|
|
392
|
+
* On 4 KiB hosts (ratio 1) this is byte-identical to clamping against the
|
|
393
|
+
* hintless default.
|
|
394
|
+
*/
|
|
395
|
+
const clampBufferPool = (bytes, pageSize = getOsPageSize()) => Math.min(scaledAnalyzePoolCap(pageSize), Math.max(ADAPTIVE_POOL_FLOOR * granuleRatio(pageSize), Math.floor(bytes)));
|
|
339
396
|
/**
|
|
340
397
|
* Buffer-pool bytes to provision per graph element (node + relationship).
|
|
341
398
|
*
|
|
@@ -354,12 +411,18 @@ const clampBufferPool = (bytes) => Math.min(defaultBufferPoolSize(), Math.max(AD
|
|
|
354
411
|
const POOL_BYTES_PER_ELEMENT = 4 * 1024;
|
|
355
412
|
/**
|
|
356
413
|
* Size the buffer pool to an estimated graph size (node + relationship count),
|
|
357
|
-
* clamped to [ADAPTIVE_POOL_FLOOR,
|
|
358
|
-
*
|
|
359
|
-
*
|
|
360
|
-
*
|
|
414
|
+
* clamped to [ADAPTIVE_POOL_FLOOR, scaledAnalyzePoolCap], with every term
|
|
415
|
+
* scaled by granuleRatio (#2631): on non-4K hosts the engine bills pool
|
|
416
|
+
* budget per OS-page-sized granule, so the same graph consumes up to
|
|
417
|
+
* pageSize/4096 × the budget it needs on x86. On 4 KiB hosts the ratio is 1
|
|
418
|
+
* and this is byte-identical to the pre-#2631 behavior. The estimate is never
|
|
419
|
+
* above the page-size-scaled #2557 cap bounded by 80% of RAM, never below the
|
|
420
|
+
* scaled COPY-safety floor; the hintless default stays unscaled.
|
|
421
|
+
*
|
|
422
|
+
* `pageSize` is a test seam (the pageSizeDoctorLines convention); production
|
|
423
|
+
* callers omit it and get the memoized real OS page size.
|
|
361
424
|
*/
|
|
362
|
-
export const estimateBufferPool = (graphElementCount) => clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT);
|
|
425
|
+
export const estimateBufferPool = (graphElementCount, pageSize = getOsPageSize()) => clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT * granuleRatio(pageSize), pageSize);
|
|
363
426
|
/**
|
|
364
427
|
* Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets
|
|
365
428
|
* it once the graph size is known (after the pipeline, before the DB open) so
|
|
@@ -394,10 +457,53 @@ const resolveBufferManagerSize = () => {
|
|
|
394
457
|
// Non-empty but unparseable input: warn the operator and fall back —
|
|
395
458
|
// mirrors the GITNEXUS_WAL_CHECKPOINT_THRESHOLD env path above.
|
|
396
459
|
if (raw.trim().length > 0) {
|
|
397
|
-
logger.warn({ rawValue: raw, fallback: defaultBufferPoolSize() }, `Ignoring invalid GITNEXUS_LBUG_BUFFER_POOL_SIZE=${raw}; expected integer >= 0 (bytes; 0 restores the native 80%-of-RAM default); falling back to
|
|
460
|
+
logger.warn({ rawValue: raw, fallback: defaultBufferPoolSize() }, `Ignoring invalid GITNEXUS_LBUG_BUFFER_POOL_SIZE=${raw}; expected integer >= 0 (bytes; 0 restores the native 80%-of-RAM default); falling back to the platform default pool size.`);
|
|
398
461
|
}
|
|
399
462
|
return defaultBufferPoolSize();
|
|
400
463
|
};
|
|
464
|
+
/**
|
|
465
|
+
* Doctor-facing view of the pool size the next Database open would get
|
|
466
|
+
* (#2631): env override > clamped hint > unscaled hintless default. Read-only;
|
|
467
|
+
* doctor prints it next to the page-size lines so support triage sees the
|
|
468
|
+
* sizing inputs at a glance. `0` is the pass-through sentinel for LadybugDB's
|
|
469
|
+
* native 80%-of-RAM default — callers must label it, not print "0 MiB".
|
|
470
|
+
*/
|
|
471
|
+
export const getEffectiveBufferPoolSize = () => resolveBufferManagerSize();
|
|
472
|
+
/**
|
|
473
|
+
* Matches the engine's buffer-pool exhaustion throw (buffer_manager.cpp:
|
|
474
|
+
* "Unable to allocate memory! The buffer pool is full and no memory could be
|
|
475
|
+
* freed!"). Distinct from isLbugPageSizeFrameError above, which matches the
|
|
476
|
+
* madvise/frame-release failure class.
|
|
477
|
+
*/
|
|
478
|
+
const BUFFER_POOL_EXHAUSTION_RE = /buffer pool is full|unable to allocate memory/i;
|
|
479
|
+
const formatMiB = (bytes) => `${Math.round(bytes / (1024 * 1024))} MiB`;
|
|
480
|
+
/**
|
|
481
|
+
* Actionable remedy for a buffer-pool exhaustion error (#2631), or undefined
|
|
482
|
+
* when `message` is not that class. Cause → consequence → remedy, the
|
|
483
|
+
* diagnoseExtensionLoad convention: names the effective pool, the override
|
|
484
|
+
* knob, and — on non-4K hosts — the granule amplification that makes the
|
|
485
|
+
* budget exhaust early (the reporter's Ascend/aarch64 64 KiB kernel billed a
|
|
486
|
+
* pool up to 16× faster than the same analyze on x86).
|
|
487
|
+
*/
|
|
488
|
+
export const bufferPoolExhaustionRemedy = (message, pageSize = getOsPageSize()) => {
|
|
489
|
+
if (!BUFFER_POOL_EXHAUSTION_RE.test(message))
|
|
490
|
+
return undefined;
|
|
491
|
+
const ratio = granuleRatio(pageSize);
|
|
492
|
+
const pool = resolveBufferManagerSize();
|
|
493
|
+
// 0 is the pass-through sentinel (GITNEXUS_LBUG_BUFFER_POOL_SIZE=0 →
|
|
494
|
+
// LadybugDB's native 80%-of-RAM default) — "0 MiB" would be nonsense in the
|
|
495
|
+
// very triage text this remedy exists to provide.
|
|
496
|
+
const poolLabel = pool === 0 ? "LadybugDB's native 80%-of-RAM default" : formatMiB(pool);
|
|
497
|
+
const pageNote = ratio > 1
|
|
498
|
+
? ` This host's ${(pageSize ?? 0) / 1024} KiB OS page size makes the engine bill pool ` +
|
|
499
|
+
`memory in ${(pageSize ?? 0) / 1024} KiB granules — up to ${ratio}× faster budget use ` +
|
|
500
|
+
`than a 4 KiB-page host running the same analyze.`
|
|
501
|
+
: '';
|
|
502
|
+
return (`The LadybugDB buffer pool (${poolLabel}) was exhausted during the bulk COPY.` +
|
|
503
|
+
pageNote +
|
|
504
|
+
` Set GITNEXUS_LBUG_BUFFER_POOL_SIZE=<bytes> to raise it (e.g. ${4 * 1024 * 1024 * 1024}` +
|
|
505
|
+
` for 4 GiB); 0 restores LadybugDB's native 80%-of-RAM default.`);
|
|
506
|
+
};
|
|
401
507
|
/** Matches WAL corruption errors from the LadybugDB engine. */
|
|
402
508
|
const WAL_CORRUPTION_RE = /corrupt(ed)?\s+wal|invalid\s+wal\s+record|wal.*corrupt|checksum.*wal/i;
|
|
403
509
|
export const WAL_RECOVERY_SUGGESTION = 'WAL corruption detected. Run `gitnexus analyze --force` to rebuild the index.';
|
|
@@ -478,8 +584,12 @@ const LBUG_PAGE_COMBO_RE = /unsupported page size combination/i;
|
|
|
478
584
|
* True when `err` looks like the LadybugDB buffer manager failing to release
|
|
479
585
|
* frame memory — the failure mode of a 4 KiB page-size assumption on a
|
|
480
586
|
* 16 KiB/64 KiB-page kernel (#1231). Deliberately does NOT match the
|
|
481
|
-
* generic "buffer pool is full" exhaustion error
|
|
482
|
-
* problem
|
|
587
|
+
* generic "buffer pool is full" exhaustion error: that one is handled as a
|
|
588
|
+
* SIZING problem — though since #2631 we know page size drives sizing too
|
|
589
|
+
* (the engine bills pool budget per OS-page-sized discard granule, so non-4K
|
|
590
|
+
* hosts exhaust the same budget up to pageSize/4096× earlier; see
|
|
591
|
+
* granuleRatio, which scales the pool accordingly, and
|
|
592
|
+
* bufferPoolExhaustionRemedy, which explains it to the operator).
|
|
483
593
|
*/
|
|
484
594
|
export const isLbugPageSizeFrameError = (err) => {
|
|
485
595
|
if (!err)
|
|
@@ -506,6 +616,15 @@ export const isPageSizeAwareLadybug = (version) => {
|
|
|
506
616
|
// `undefined` = not probed yet; `null` = probed and unavailable. Cached
|
|
507
617
|
// because analyze error paths and doctor may both ask, and getconf forks.
|
|
508
618
|
let cachedOsPageSize;
|
|
619
|
+
/**
|
|
620
|
+
* Test seam (the `_captureLogger` convention): pin the memoized OS page size
|
|
621
|
+
* so sizing tests are host-independent — without this they would silently
|
|
622
|
+
* drift on 16 KiB-page Apple Silicon runners. `number` pins a value, `null`
|
|
623
|
+
* pins "undetectable", `undefined` clears the memo so the next call re-probes.
|
|
624
|
+
*/
|
|
625
|
+
export const _setOsPageSizeForTests = (pageSize) => {
|
|
626
|
+
cachedOsPageSize = pageSize;
|
|
627
|
+
};
|
|
509
628
|
/**
|
|
510
629
|
* OS memory page size in bytes, or `undefined` when it cannot be determined
|
|
511
630
|
* (Windows, missing getconf, sandboxed exec). Node exposes no page-size API,
|
|
@@ -545,10 +664,6 @@ export const getOsPageSize = () => {
|
|
|
545
664
|
}
|
|
546
665
|
return cachedOsPageSize ?? undefined;
|
|
547
666
|
};
|
|
548
|
-
/** Exported only for unit tests — clears the getconf probe cache. */
|
|
549
|
-
export const _resetOsPageSizeCacheForTest = () => {
|
|
550
|
-
cachedOsPageSize = undefined;
|
|
551
|
-
};
|
|
552
667
|
/**
|
|
553
668
|
* Return true when the error message indicates that a LadybugDB write
|
|
554
669
|
* transaction could not proceed due to lock contention — either a file
|
package/package.json
CHANGED